<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">887894</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.887894</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Anticancer Peptide Prediction <italic>via</italic> Multi-Kernel CNN and Attention Model</article-title>
<alt-title alt-title-type="left-running-head">Wu et al.</alt-title>
<alt-title alt-title-type="right-running-head">Anticancer Peptide Prediction <italic>via</italic> Multi-CNN and Attention Model</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Xiujin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1632729/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zeng</surname>
<given-names>Wenhua</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lin</surname>
<given-names>Fan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/865944/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Peng</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Xinzhu</given-names>
</name>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Informatics</institution>, <institution>Xiamen University</institution>, <addr-line>Xiamen</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Boston Children&#x2019;s Hospital</institution>, <addr-line>Boston</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Chongqing Michong Technology Co., Ltd.</institution>, <addr-line>Chongqing</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1064518/overview">Leyi Wei</ext-link>, Shandong University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/632487/overview">Min Wu</ext-link>, Institute for Infocomm Research (A&#x2217;STAR), Singapore</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1717829/overview">Nanqing Liu</ext-link>, Southwest Jiaotong University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Fan Lin, <email>iamafan@xmu.edu.cn</email>; Wenhua Zeng, <email>whzeng@xmu.edu.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>27</day>
<month>04</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>887894</elocation-id>
<history>
<date date-type="received">
<day>02</day>
<month>03</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>25</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Wu, Zeng, Lin, Xu and Li.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Wu, Zeng, Lin, Xu and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Background:</bold> Modern lifestyles mean that people are more likely to suffer from some form of cancer. As anticancer peptides can effectively kill cancer cells and play an important role in fighting cancer, they have been a subject of increasing research interest.</p>
<p>
<bold>Methods:</bold> This study presents a useful tool to identify the anticancer peptides based on a multi-kernel CNN and attention model, called ACP-MCAM. This model can automatically learn adaptive embedding and the context sequence features of ACP. In addition, to obtain better interpretability and integrity, we visualized the model.</p>
<p>
<bold>Results:</bold> Benchmarking comparison shows that ACP-MCAM significantly outperforms several state-of-the-art models. Different encoding schemes have different impacts on the performance of the model. We also studied tmethod parameter optimization.</p>
<p>
<bold>Conclusion:</bold> The ACP-MCAM can integrate multi-kernel CNN and self-attention mechanism, which outperforms the previous model in identifying anticancer peptides. It is expected that the work will provide new research ideas for anticancer peptide prediction in the future. In addition, this work will promote the development of the interdisciplinary field of artificial intelligence and biomedicine.</p>
</abstract>
<kwd-group>
<kwd>anticancer peptide</kwd>
<kwd>multi-CNN</kwd>
<kwd>attention mechanism</kwd>
<kwd>prediction</kwd>
<kwd>classfication</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Anticancer peptide (<xref ref-type="bibr" rid="B22">Plumb et al., 2019</xref>) (ACP) is a polypeptide sequence with anticancer activity. It is composed of 10&#x2013;50 amino acid amino acids (<xref ref-type="bibr" rid="B4">Barras and Widmann, 2011</xref>). Its molecular structure is complex. It is a molecular polymer between amino acids and proteins (<xref ref-type="bibr" rid="B19">Li et al., 2006</xref>), which is composed of several or dozen amino acids connected by peptide bonds (<xref ref-type="bibr" rid="B8">Gaspar et al., 2013</xref>). It can destroy the structure of the tumor cell membrane and inhibit the proliferation and migration of cancer cells (<xref ref-type="bibr" rid="B28">Song et al., 2020</xref>). It can induce the apoptosis of cancer cells without damage to normal human cells (<xref ref-type="bibr" rid="B23">Qiao et al., 2019</xref>; <xref ref-type="bibr" rid="B26">Ryu et al., 2021</xref>). At present, most anti-cancer drugs have side effects on the kidney (<xref ref-type="bibr" rid="B12">Kamisli et al., 2015</xref>; <xref ref-type="bibr" rid="B32">Van Acker et al., 2016</xref>), nerve and heart (<xref ref-type="bibr" rid="B22">Plumb et al., 2019</xref>), and gonads (<xref ref-type="bibr" rid="B21">Novin and Sciences, 2014</xref>; <xref ref-type="bibr" rid="B9">Gutierrez et al., 2016</xref>). Compared with conventional chemotherapy, anticancer peptides have the advantages of high specificity, low production cost, high tumor penetration rate, and easy synthesis and modification (<xref ref-type="bibr" rid="B11">Otvos, 2008</xref>). Due to the benefits of anti-cancer peptides, more and more anti-cancer peptides are used in clinical trials. For example, three peptides Didemin A, B, and C, which are extracted from sea squirt, have obvious inhibitory effects on breast cancer, ovarian cancer, and uterine cancer, and have entered phase II clinical trials (<xref ref-type="bibr" rid="B6">Clamp and Jayson, 2002</xref>). The synthetic peptide Elisidepsin (PM02734) has also entered phase II clinical trials (<xref ref-type="bibr" rid="B25">Ratain et al., 2015</xref>). Identifying anti-cancer peptides is of great significance for discovering new and efficient treatments for diseases. Traditional anti-cancer peptide prediction relies on biological experiments, and the prediction is accurate, but it is inefficient, time-consuming, and costly. With the development of Qualcomm sequencing technology, protein sequence data is increasing exponentially every year, and massive sequence data presents severe challenges to biological experiment technology (<xref ref-type="bibr" rid="B25">Ratain et al., 2015</xref>).</p>
<p>Some research methods have used machine learning and deep learning methods to build anticancer peptide prediction models (<xref ref-type="bibr" rid="B30">Su et al., 2019</xref>; <xref ref-type="bibr" rid="B29">Su et al., 2020</xref>; <xref ref-type="bibr" rid="B36">Wei et al., 2020</xref>). Tyagi et al. proposed an anti-cancer peptide prediction method called AntiCP (<xref ref-type="bibr" rid="B31">Tyagi et al., 2013</xref>), which uses amino acid composition, dipeptide composition, and composition differences between amino acid N-terminus and C-terminus, combined with a support vector machine (SVM) model for prediction. Chen et al. (<xref ref-type="bibr" rid="B5">Chen et al., 2016</xref>) used the features of amino acid dipeptide composition and pseudo-amino acid composition, combined with a support vector machine to construct an anti-cancer peptide prediction algorithm called iACP (<xref ref-type="bibr" rid="B5">Chen et al., 2016</xref>). Wei et al. have proposed the ACPred-FL (<xref ref-type="bibr" rid="B15">Leyi et al., 2018</xref>) model and the PEPred-Suite model (<xref ref-type="bibr" rid="B16">Leyi et al., 2019</xref>). The ACPred-FL (<xref ref-type="bibr" rid="B15">Leyi et al., 2018</xref>) model used four sequence feature representation samples: binary profile features (BPF), G-gap dipeptide composition (GDC), overlapping property (OPF), and composition transition distribution (CTD), and the frequency of each amino acid in the sequence, combined with the SVM model to build 40 sub-models, and then the output of the 40 sub-models are used as the input feature to build the model for anti-cancer peptide prediction. AntiCP 2.0 (<xref ref-type="bibr" rid="B1">Agrawal et al., 2020</xref>) uses SVM, ETree, random forest, ridge algorithm, artificial neural network (ANN), and the K nearest neighbor (KNN) method to construct the prediction model of anticancer peptides. There have also been integrated anti-cancer peptide prediction methods, which integrate multiple or multiple machine learning methods to predict various peptide sequences (<xref ref-type="bibr" rid="B18">Li et al., 2017</xref>; <xref ref-type="bibr" rid="B37">Wei et al., 2017</xref>; <xref ref-type="bibr" rid="B15">Leyi et al., 2018</xref>).</p>
<p>Existing models have some problems, such as low recognition accuracy, insufficient generalization ability, and there is a lack of large-scale evaluation of features and prediction models. Almost all the existing anticancer peptide prediction studies use sequence features to construct anticancer peptide prediction models, which show that the anticancer peptide prediction methods based on sequence information are effective. But most research to date has not considered the combination of protein structure and sequence data characteristics or the feature space used was not comprehensive enough. Most of them use single machine learning methods, such as support vector machines, and seldom use the attention mechanism (<xref ref-type="bibr" rid="B33">Vaswani et al., 2017</xref>). To solve the above problems, this article takes the anti-cancer peptide sequence data as the research object, exploring the anti-cancer peptide prediction method based on the attention mechanism (<xref ref-type="bibr" rid="B33">Vaswani et al., 2017</xref>) and deep learning models, establishing a relatively advanced and effective anti-cancer peptide prediction model. The main research contents are summarized as follows:<list list-type="simple">
<list-item>
<p>1) The paper proposes a model with strong anti-cancer peptide recognition ability based on learnable adaptive embedding and amino acid structure features. It is a self-attention mechanism that can automatically learn the context sequence features of ACP, learns the contribution of each amino acid node in the entire anti-cancer peptide sequence, automatically captures the global information in the ACP sequence, and can capture the contributions of protein cluster formed by 3&#x2013;5 amino acid nodes in the anti-cancer peptide sequence to improve the ability to identify the anti-cancer peptide model.</p>
</list-item>
<list-item>
<p>2) This article comprehensively evaluates the different feature projects of anti-cancer peptides, constructs a deep learning model with strong performance, and integrates the advantages of multiple deep learning models to improve predictive performance. To improve the interpretability of the model prediction results, this article visualizes model prediction characteristics to improve the interpretability of the model prediction results.</p>
</list-item>
<list-item>
<p>3) The methods for identifying various functional peptides of the same type from the same functional peptides are relatively similar. It is difficult to identify anti-cancer peptides from multiple peptide sequences. Most of the existing methods recognize anti-cancer peptides from a single type of peptide sequence, so research on new deep learning models is needed. The new deep learning model studies the impact of different coding schemes on the performance of the model, examining the method of parameter optimization to build a better deep learning model and obtain the best model for identifying anticancer peptides from different functional peptides.</p>
</list-item>
</list>
</p>
<p>This article first examines the peptide datasets used, which are introduced in <italic>Datasets Section</italic>. Second, the anticancer peptide predicting model of multi-kernel CNN and the attention mechanism is explained in detail in <italic>Model Overview Section</italic>. Third, the performance evaluation index, loss function, experimental process, and results of the model are presented in <italic>Experiments and Results Section</italic>. Finally, the results and the prospects for future work are discussed and summarized in <italic>Conclusion Section</italic>.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methods</title>
<p>This chapter introduces the structural frame of the ACP-MCAM model and the datasets of anti-cancer peptides in detail.</p>
<sec id="s2-1">
<title>Datasets</title>
<p>The datasets used in this experiment come from the literature (<xref ref-type="bibr" rid="B15">Leyi et al., 2018</xref>). The samples in this dataset were also collected from literature (<xref ref-type="bibr" rid="B31">Tyagi et al., 2013</xref>; <xref ref-type="bibr" rid="B2">Atul et al., 2015</xref>). Among them, the positive samples are anti-cancer peptides that have been confirmed by physical experiments, and the negative samples are selected from anti-microbial peptides that have no anti-cancer activity. The training dataset ACPs500 consists of 250 anti-cancer peptides and 250 non-anti-cancer peptides. The test dataset ACPs164 consists of 82 positive samples and 82 non-anticancer peptide samples. All of the samples are filtered by CD-HIT (<xref ref-type="bibr" rid="B38">Ying et al., 2010</xref>) to filter out redundant sequences with a similarity higher than 90% so that the sequences in the training dataset and the test dataset are different. Two other data sets, neuropeptides and antifungal peptides, were also used in this paper. The details of the dataset are shown in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary of datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Datasets</th>
<th align="center">Dataset Type</th>
<th align="center">Total Number</th>
<th align="center">Number of Positive Samples</th>
<th align="center">Number of Negative Samples</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">ACPs500</td>
<td align="left">Training set</td>
<td align="char" char=".">500</td>
<td align="char" char=".">250</td>
<td align="char" char=".">250</td>
</tr>
<tr>
<td align="left">ACPs164</td>
<td align="left">Test set</td>
<td align="char" char=".">164</td>
<td align="char" char=".">82</td>
<td align="char" char=".">82</td>
</tr>
<tr>
<td align="left">NPs1400</td>
<td align="left">Training set</td>
<td align="char" char=".">1400</td>
<td align="char" char=".">700</td>
<td align="char" char=".">700</td>
</tr>
<tr>
<td align="left">NPs350</td>
<td align="left">Test set</td>
<td align="char" char=".">350</td>
<td align="char" char=".">175</td>
<td align="char" char=".">175</td>
</tr>
<tr>
<td align="left">AFPs2336</td>
<td align="left">Training set</td>
<td align="char" char=".">2336</td>
<td align="char" char=".">1168</td>
<td align="char" char=".">1168</td>
</tr>
<tr>
<td align="left">AFPs582</td>
<td align="left">Test set</td>
<td align="char" char=".">582</td>
<td align="char" char=".">291</td>
<td align="char" char=".">291</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-2">
<title>Model Overview</title>
<p>This chapter details the ACP-MCAM model structural frame used to predict the anticancer peptides.</p>
<p>The architecture of the ACP-MCAM model is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. It consists of five modules, 1) an embedding layer, 2) a Multi-kernel CNN layer, 3) the position embedding layer, 4) the encoder layer, and 5) the task output layer. In module 1), the embedding layer first processes the input anticancer peptide sequence and converts each amino acid of the anticancer peptide sequence into a low dimensional dense vector as the embedding vector representation of amino acid nodes. No matter where the amino acid appears in the sequence, the same type of amino acid uniquely corresponds to the same vector. 2) In the multi-kernel CNN layer, this paper uses convolution neural network (CNN) technology to encode the amino acid nodes of anticancer peptide sequence by using the context information and different semantic information of specific amino acids in the anticancer peptide sequence. We perform a two-dimensional convolution operation with padding on the output of the embedding layer to ensure that the dimensions of the input and output are the same. The kernel can take odd numbers such as 1, 3, and 5, connect them in the last dimension, and then do a linear transformation. 3) The position embedding layer encodes the position information of the amino acids in the anti-cancer peptide sequence. It is a vector containing the position embedding information of the amino acid sequence. 4) The encoding layer is the core of the model, and the input feature matrix is the output of the position embedding layer and the multi-kernel CNN layer. The encoder layer includes multiple encoder blocks. Each encoder block is based on a multi-head attention mechanism and a fully connected neural network. The feed forward part of each encoder block ensures that the input and output sizes of each encoder block are consistent. The sequential stacking of multiple coding layers makes the representation of anticancer peptide sequences more effective. In module 4), the encoder layer is used to capture the context of each remaining embedding vector at different positions, so that the remaining embedding has different feature vectors according to the context, and learning the discriminative features of ACP. Finally, module 5) is the last part of the model, called the task output layer, which is composed of a fully connected neural network and a nonlinear activation function. It converts the representation of ACPs into the probability distribution of the classes for prediction. Please note that the penultimate neural network in the task output layer is specifically designed for feature visualization. These five modules are described in detail below.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The framework of the proposed ACP-MCAM. <bold>(A)</bold> Embedding layer. <bold>(B)</bold> Multi-kernel CNN layer. <bold>(C)</bold> Position embedding layer. <bold>(D)</bold> Encoding layer. <bold>(E)</bold> Task-output layer.</p>
</caption>
<graphic xlink:href="fgene-13-887894-g001.tif"/>
</fig>
<sec id="s2-2-1">
<title>Embedding Layer</title>
<p>The core idea of embedding is to map all of the amino acids in the anticancer peptide sequence into a dense vector in a low-dimensional space (mostly <italic>K</italic> &#x3d; 50&#x2013;300 dimensions). Since the embedding layer maps each amino acid into a K-dimensional vector. If there are n amino acids in all of the anti-cancer peptide sequences. All of the peptide sequences can be represented by an <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> dimensional matrix. In this article, the anticancer peptide sequences are consist of 20 different amino acids (&#x27;A&#x27;,&#x27;C&#x27;,&#x27;D&#x27;,&#x27;E&#x27;,&#x27;F&#x27;,&#x27;G&#x27;,&#x27;H&#x27;,&#x27;I&#x27;,&#x27;K&#x27;,&#x27;L&#x27;,&#x27;M&#x27;,&#x27;N&#x27;,&#x27;P&#x27;,&#x27;Q&#x27;,&#x27;R&#x27;,&#x27;S&#x27;,&#x27;T&#x27;,&#x27;V&#x27;,&#x27;W&#x27;,&#x27;Y&#x27;). It is not enough to use the underlying embedding feature as the representative feature of the original peptide sequence, a higher level feature needs to be processed on this basis.</p>
</sec>
<sec id="s2-2-2">
<title>Single CNN</title>
<p>Modern convolutional neural networks were proposed by LeCun (<xref ref-type="bibr" rid="B14">Lecun et al., 1998</xref>). They show excellent performance in solving computer vision problems such as image classification, recognition, and understanding (<xref ref-type="bibr" rid="B7">Farabet et al., 2013</xref>; <xref ref-type="bibr" rid="B35">Wang et al., 2017</xref>; <xref ref-type="bibr" rid="B39">Yu et al., 2017</xref>). The chief conception of the convolutional neural network (CNN) is to capture the local features of the object. At first, it achieved great success in the image field, and later it has also been widely used in the text field. For the anti-cancer peptide data, the local feature is the sliding window composed of several amino acids which is similar to N-gram. The advantage of the convolutional neural network is that it can automatically combine and filter features to obtain semantic information at a different level. For an anticancer peptide sequence &#x201c;GATCDCPLR&#x201d;, if kernel &#x3d; n, the features of n consecutive amino acids on the anticancer peptide sequence can be extracted by convoluting the amino acid of the anticancer peptide sequence. Different kernels can gain different combinations of semantic information of the amino acid in the anticancer peptide sequences. Since each step of the convolution uses the weight sharing mechanism, the training speed is relatively fast. In this article, we use the weight sharing mechanism of CNN to extract the feature of anticancer peptides which have achieved good effects.</p>
</sec>
<sec id="s2-2-3">
<title>Multi-Kernel CNN Layer</title>
<p>The Multi-kernel CNN layer mainly connects multiple convolution kernels of different lengths and combines different anticancer peptide amino acids to obtain different semantic information. For example, for an anti-cancer peptide sequence &#x201c;GATCDCPLR&#x201d;, if kernel &#x3d; 3, the amino acid letter &#x201c;C&#x201d; on the anti-cancer peptide sequence is convoluted, and the result of convolution features with sky blue color can be obtained, which is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. If kernel &#x3d; 5, the amino acid letter &#x201c;C&#x201d; on the anti-cancer peptide sequence is convoluted, and the result of convolution features with yellow color can be obtained, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. We concatenate these two convolution results to obtain the convolution features in the first dimension. Convolution kernels of different lengths act on the output matrix of the anti-cancer peptide sequence embedding layer, which can capture different semantic length information and combine them as the input of the deep network.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Multi-kernel CNN to extract anticancer peptide sequence features.</p>
</caption>
<graphic xlink:href="fgene-13-887894-g002.tif"/>
</fig>
</sec>
<sec id="s2-2-4">
<title>Position Embedding Layer</title>
<p>The original input of the model is an embedding vector without the position information of the amino acid, and the position encoding layer combines the position information with the amino acid embedding vector to form new features and then input into the model. If an anti-cancer peptide sequence of length n is input, the encoding mode of formula (1) (2) can output a unique position code for each time step. Moreover, the distance between any two time steps between anticancer peptide sequences with different lengths is the same.<disp-formula id="e1">
<mml:math id="m2">
<mml:mrow>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">sin</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>10000</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">mod</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
<disp-formula id="e2">
<mml:math id="m3">
<mml:mrow>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:msub>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">cos</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>10000</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">mod</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>Among them, position embedding (PE) indicates the code containing the specific position information of the anticancer peptide after being coded. P represents the position of the amino acid in the sequence, <inline-formula id="inf2">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">mod</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the dimension of the position vector, and <inline-formula id="inf3">
<mml:math id="m5">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">mod</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the <italic>i</italic>th dimension of the position <inline-formula id="inf4">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">mod</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> dimensional position vector. So according to the above formula, we can get the position embedding vector of the <italic>p</italic>th amino acid.</p>
</sec>
<sec id="s2-2-5">
<title>Encoding Layer</title>
<p>The basic module of the encoder layer is the encoder model of the transformer (<xref ref-type="bibr" rid="B33">Vaswani et al., 2017</xref>). Each encode block includes a multi-head attention mechanism, a feed forward network, and two residual connections. Multi-head attention consists of several self-attention mechanisms, which are used to learn the contextual representation of the sequence. If there are 3 heads, then we linearly transform the features of the anticancer peptide sequence to get the query vector (q1, q2, q3), key vector (k1, k2, k3), and value vector (v1, v2, v3). The query and key calculate the correlation score, named attention score, and then the value is weighted and summed according to the attention score as shown in formula (3) (4).<disp-formula id="e3">
<mml:math id="m7">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="italic">Q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">X</mml:mi>
<mml:msup>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mi mathvariant="italic">Q</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="italic">K</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">X</mml:mi>
<mml:msup>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mi mathvariant="italic">K</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="italic">V</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">X</mml:mi>
<mml:msup>
<mml:mi mathvariant="italic">W</mml:mi>
<mml:mi mathvariant="italic">V</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m8">
<mml:mrow>
<mml:mi mathvariant="italic">A</mml:mi>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mi mathvariant="italic">e</mml:mi>
<mml:mi mathvariant="italic">n</mml:mi>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mi mathvariant="italic">o</mml:mi>
<mml:mi mathvariant="italic">n</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="italic">Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="italic">K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="italic">V</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="italic">o</mml:mi>
<mml:mi mathvariant="italic">f</mml:mi>
<mml:mi mathvariant="italic">t</mml:mi>
<mml:mi mathvariant="italic">max</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">Q</mml:mi>
<mml:msup>
<mml:mi mathvariant="italic">K</mml:mi>
<mml:mi mathvariant="italic">T</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">d</mml:mi>
<mml:mi mathvariant="italic">k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="italic">V</mml:mi>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>Suppose we now want to know the attention score of the first amino acid node &#x201c;G" in the anticancer peptide sequence. We calculate the respective attention score by the embedding vector of the amino acid node &#x201c;G" and the embedding vectors of all of the other amino acid nodes in the anticancer peptide sequence. These scores determine the attention weight of the amino acid &#x201c;G" node embedding vector when we encode the node &#x201c;G".</p>
<p>In the first step, the query, key, and value vectors are obtained by multiplying the anticancer peptide embedding vector with three parameter matrices. These three parameter matrices are the parameters that the model needs to learn. The second step is to multiply Q by the transpose of K to obtain an initial attention score. Then divide each score by <inline-formula id="inf5">
<mml:math id="m9">
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf6">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the dimension of the key vector which is used to make the model more stable when calculating the gradient when backpropagating. The third step is to pass these scores through a softmax function. The softmax function can normalize the scores into probability representation.</p>
</sec>
<sec id="s2-2-6">
<title>Task-Output Layer</title>
<p>The output vector of the encoder layer is the feature of anticancer peptides. The main job of the task-output layer is to convert the output vector to binary classification. The task-output layer mainly contains several important modules: linear connection layer, residual network layer, normalization layer, and activation function.</p>
<p>The linear connection layer is a fully connected neural network. It obtains the output of the specified dimension through the linear change of the previous step and plays the role of transforming the dimension. The final dimension corresponds to the number of output categories.</p>
<p>The residual network layer is implemented in the form of skip layer connections, and the input unit is directly added to the output unit. The residual network can solve the degradation problem of the deep neural network well. The residual network converges faster than the same number of layers.</p>
<p>The normalization layer is a standard network layer required by the deep network model. As the number of network layers increases, the output value will become too large or too small. It may cause abnormal and the model may converge very slowly. The normalization layer is used to normalize the output value. Then the output value can be in a reasonable range.</p>
<p>The main role of the activation function is to provide the nonlinear modeling capability of the network. To avoid the pure linear combination, we add an activation function (tanh, ReLU, Softmax, etc.) after the output of each layer. ReLU can keep the gradient undecayed when x &#x3e; 0, thereby alleviating the problem of gradient disappearance. The output mean of tanh is 0, and its convergence speed is fast, which can reduce the number of iterations. Using different combinations of activation functions can make the network achieve better results.</p>
</sec>
<sec id="s2-2-7">
<title>Focal Loss Function</title>
<p>Focal loss (FL) is mainly a strategy proposed by Kaiming to solve the classification problem of indistinguishable samples, that is, to set weights according to the contribution of difficult and easy samples to the loss. The formula is as follows:<disp-formula id="e5">
<mml:math id="m11">
<mml:mrow>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">p</mml:mi>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">&#x3b1;</mml:mi>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">p</mml:mi>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi mathvariant="normal">&#x3b3;</mml:mi>
</mml:msup>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="normal">log</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">p</mml:mi>
<mml:mi mathvariant="italic">t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where, <inline-formula id="inf7">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf8">
<mml:math id="m13">
<mml:mtext>&#x3b3;</mml:mtext>
</mml:math>
</inline-formula> are parameters that are used to coordinate control samples that are difficult to distinguish, and <inline-formula id="inf9">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> represents the probability of ground-truth class. The easier the sample is to distinguish, the larger <inline-formula id="inf10">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is, and the smaller contribution to the loss. Conversely, the greater the loss of the hard-to-separate sample. This article uses the parameter <inline-formula id="inf11">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b1;</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.3</mml:mn>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mtext>&#xa0;&#x3b3;</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> setting to achieve the best results.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<title>Experiments and Results</title>
<p>In this section, the performance of the ACP-MCAM model is evaluated in several evaluation metrics. We compare it with other models and discuss the results. The experimental parameters are then discussed to verify the effectiveness of the model.</p>
<sec id="s3-1">
<title>Evaluating Metrics</title>
<p>The following evaluation indicators were used to evaluate our model. Including recall, precision, accuracy (ACC), Matthew correlation coefficient (MCC), and the area under the ROC curve (AUC) (<xref ref-type="bibr" rid="B3">Balachandran et al., 2018</xref>; <xref ref-type="bibr" rid="B17">Leyi et al., 2020</xref>; <xref ref-type="bibr" rid="B20">Mehedi et al., 2020</xref>). The specific formula is as follows:<disp-formula id="e6">
<mml:math id="m17">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">Re</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">f</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">Pr</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">p</mml:mi>
<mml:mi mathvariant="italic">r</mml:mi>
<mml:mi mathvariant="italic">e</mml:mi>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mi mathvariant="italic">o</mml:mi>
<mml:mi mathvariant="italic">n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="italic">r</mml:mi>
<mml:mi mathvariant="italic">e</mml:mi>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mi mathvariant="italic">a</mml:mi>
<mml:mi mathvariant="italic">l</mml:mi>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">p</mml:mi>
<mml:mi mathvariant="italic">r</mml:mi>
<mml:mi mathvariant="italic">e</mml:mi>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mi mathvariant="italic">s</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
<mml:mi mathvariant="italic">o</mml:mi>
<mml:mi mathvariant="italic">n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="italic">r</mml:mi>
<mml:mi mathvariant="italic">e</mml:mi>
<mml:mi mathvariant="italic">c</mml:mi>
<mml:mi mathvariant="italic">a</mml:mi>
<mml:mi mathvariant="italic">l</mml:mi>
<mml:mi mathvariant="italic">l</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>Among them, true positive (TP) and false negative (FN) represent the number of true anticancer peptides that are predicted correctly and incorrectly. True negative (TN) and false positive (FP) represent the number of non-anticancer peptides that are predicted correctly and incorrectly. Accuracy is the percentage of correctly classified samples in all samples. The sensitivity (SE) and specificity (SP) index measure the predictive ability of predictors for positive and negative samples, respectively. Precision represents the prediction success rate of positive samples. The recall represents the proportion of predicting positive samples in all true positive samples. The F1 score is the coordinated average value of accuracy and recall. The higher the selected value, the better the performance of the model. The other two indicators, AUC and MCC measure the overall performance of the predictor. AUC sorts all samples obtained from model evaluation by score. By calculating the area enclosed by the ROC curve, the AUC value can be obtained.</p>
</sec>
<sec id="s3-2">
<title>Comparison Between ACP-MCAM and Existing Models in Ten-fold Cross Validation</title>
<p>To verify the predictive performance of anticancer peptide ACP-MCAM, we compared it with several existing models, including iACP(<xref ref-type="bibr" rid="B5">Chen et al., 2016</xref>), ACPred-FL (<xref ref-type="bibr" rid="B15">Leyi et al., 2018</xref>), PEPred-Suite (<xref ref-type="bibr" rid="B16">Leyi et al., 2019</xref>), ACPred-Fuse (<xref ref-type="bibr" rid="B24">Rao et al., 2020</xref>), AntiCP_ACC (<xref ref-type="bibr" rid="B34">Vijayakumar and Ptv, 2015</xref>), AntiCP_DC(<xref ref-type="bibr" rid="B34">Vijayakumar and Ptv, 2015</xref>) and Hajisharifi&#x2019;s(<xref ref-type="bibr" rid="B10">Hajisharifi et al., 2014</xref>). The cross validation results are shown in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Cross validation results of ACP-MCAM and existing models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Methods</th>
<th align="center">SE (%)</th>
<th align="center">SP (%)</th>
<th align="center">Accuracy (%)</th>
<th align="center">MCC (%)</th>
<th align="center">AUC (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">iACP</td>
<td align="char" char=".">57.2</td>
<td align="char" char=".">84.0</td>
<td align="char" char=".">70.6</td>
<td align="char" char=".">42.8</td>
<td align="char" char=".">80.9</td>
</tr>
<tr>
<td align="left">ACPred-FL</td>
<td align="char" char=".">71.6</td>
<td align="char" char=".">84.4</td>
<td align="char" char=".">78.0</td>
<td align="char" char=".">56.5</td>
<td align="char" char=".">84.6</td>
</tr>
<tr>
<td align="left">PEPred-Suite</td>
<td align="char" char=".">72.8</td>
<td align="char" char=".">88.0</td>
<td align="char" char=".">80.4</td>
<td align="char" char=".">61.5</td>
<td align="char" char=".">86.0</td>
</tr>
<tr>
<td align="left">ACPred-Fuse</td>
<td align="char" char=".">77.2</td>
<td align="char" char=".">87.6</td>
<td align="char" char=".">82.4</td>
<td align="char" char=".">65.2</td>
<td align="char" char=".">88.2</td>
</tr>
<tr>
<td align="left">AntiCP_ACC</td>
<td align="char" char=".">66.8</td>
<td align="char" char=".">78.4</td>
<td align="char" char=".">72.6</td>
<td align="char" char=".">45.5</td>
<td align="char" char=".">82.4</td>
</tr>
<tr>
<td align="left">AntiCP_DC</td>
<td align="char" char=".">71.6</td>
<td align="char" char=".">77.6</td>
<td align="char" char=".">74.6</td>
<td align="char" char=".">49.3</td>
<td align="char" char=".">82.5</td>
</tr>
<tr>
<td align="left">Hajisharifi&#x2019;s</td>
<td align="char" char=".">67.2</td>
<td align="char" char=".">83.6</td>
<td align="char" char=".">75.4</td>
<td align="char" char=".">51.5</td>
<td align="char" char=".">83.1</td>
</tr>
<tr>
<td align="left">ACP-MCAM</td>
<td align="char" char=".">
<bold>85.6</bold>
</td>
<td align="char" char=".">
<bold>95.2</bold>
</td>
<td align="char" char=".">
<bold>90.4</bold>
</td>
<td align="char" char=".">
<bold>81.3</bold>
</td>
<td align="char" char=".">
<bold>91.9</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: The best results are marked in bold and the second best results are underlined.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="T2">Table 2</xref>, we can see that the performance of our proposed method ACP-MCAM on all indicators (SE, SP, Accuracy, MCC, and AUC) is significantly better than other predictors, reaching 85.6, 95.2, 90.4, 81.3, and 91.9%, respectively. SE, SP, Accuracy, and MCC are 8.4, 7.8, 8, and 16.1%, which is higher than other predictors.</p>
</sec>
<sec id="s3-3">
<title>Comparison Between ACP-MCAM and Existing Models in Independent Test</title>
<p>In order to verify the superiority of the proposed ACP-MCAM model, we used an independent test dataset to compare its performance with several existing predictions. As shown in <xref ref-type="table" rid="T3">Table 3</xref>, we can see that the performance of our proposed method ACP-MCAM on all indicators is significantly better than other predictors. SE&#x3001;SP&#x3001;ACC&#x3001;MCC&#x548c;AUC have reached 85.4, 96.3, 90.9, 82.2 and 94.8%, respectively. Especially SE, MCC, and AUC are 13.4, 50.2, and 8% higher than other predictors. In general, independent test results confirm that our prediction method performs better than other prediction methods, and can better distinguish true anti-cancer peptides from non-anti-cancer peptides.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Independent test results of ACP-MCAM and existing models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Methods</th>
<th align="center">SE (%)</th>
<th align="center">SP (%)</th>
<th align="center">Accuracy (%)</th>
<th align="center">MCC (%)</th>
<th align="center">AUC (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">iACP</td>
<td align="char" char=".">54.9</td>
<td align="char" char=".">88.8</td>
<td align="char" char=".">87.7</td>
<td align="char" char=".">22.6</td>
<td align="char" char=".">76.1</td>
</tr>
<tr>
<td align="left">ACPred-FL</td>
<td align="char" char=".">69.5</td>
<td align="char" char=".">85.8</td>
<td align="char" char=".">85.3</td>
<td align="char" char=".">25.9</td>
<td align="char" char=".">85.1</td>
</tr>
<tr>
<td align="left">PEPred-Suite</td>
<td align="char" char=".">68.3</td>
<td align="char" char=".">90.6</td>
<td align="char" char=".">89.9</td>
<td align="char" char=".">32.0</td>
<td align="char" char=".">86.1</td>
</tr>
<tr>
<td align="left">ACPred-Fuse</td>
<td align="char" char=".">72</td>
<td align="char" char=".">89.5</td>
<td align="char" char=".">89</td>
<td align="char" char=".">32.0</td>
<td align="char" char=".">86.8</td>
</tr>
<tr>
<td align="left">AntiCP_ACC</td>
<td align="char" char=".">68.3</td>
<td align="char" char=".">88.5</td>
<td align="char" char=".">87.9</td>
<td align="char" char=".">28.8</td>
<td align="char" char=".">85.3</td>
</tr>
<tr>
<td align="left">AntiCP_DC</td>
<td align="char" char=".">68.3</td>
<td align="char" char=".">82.6</td>
<td align="char" char=".">82.2</td>
<td align="char" char=".">22.3</td>
<td align="char" char=".">83.0</td>
</tr>
<tr>
<td align="left">Hajisharifi&#x2019;s</td>
<td align="char" char=".">69.5</td>
<td align="char" char=".">88.4</td>
<td align="char" char=".">87.9</td>
<td align="char" char=".">29.2</td>
<td align="char" char=".">85.5</td>
</tr>
<tr>
<td align="left">ACP-MCAM</td>
<td align="char" char=".">
<bold>85.4</bold>
</td>
<td align="char" char=".">
<bold>96.3</bold>
</td>
<td align="char" char=".">
<bold>90.9</bold>
</td>
<td align="char" char=".">
<bold>82.2</bold>
</td>
<td align="char" char=".">
<bold>94.8</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: The best results are marked in bold and the second best results are underlined.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-4">
<title>Parameter Analysis</title>
<p>Several important parameters may affect the performance of our models, such as the learning rate and the kernel of multi-kernel CNN. Learning rate is an important parameter of deep learning. Through the adjustment of learning rate, we can see whether the objective function can quickly converge to the minimum value and fall into the local optimal value. An appropriate learning rate can make the objective function converge to the optimal value quickly.</p>
<p>In this section, we will perform a sensitivity analysis on these parameters. In our model, the number of training epochs is set to 50. The output dimension is 64. We train our model by modifying the learning rate. <xref ref-type="table" rid="T4">Table 4</xref> shows that as the learning rate changes, the performance first gradually increases and then decreases. If the learning rate is equal to 2e-4, three of the five main evaluation indexes are the best. Accuracy, AUC, and F1-score are the highest. The model has achieved the best performance.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>The performance of the ACP-MCAM model affected by the learning rate.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Learning Rate</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1-Score</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1e-4</td>
<td align="char" char=".">0.8292</td>
<td align="char" char=".">0.8</td>
<td align="char" char=".">0.878</td>
<td align="char" char=".">0.8372</td>
<td align="char" char=".">0.9225</td>
</tr>
<tr>
<td align="left">2e-4</td>
<td align="char" char=".">0.9085</td>
<td align="char" char=".">0.9589</td>
<td align="char" char=".">0.8536</td>
<td align="char" char=".">0.9032</td>
<td align="char" char=".">0.9479</td>
</tr>
<tr>
<td align="left">3e-4</td>
<td align="char" char=".">0.8841</td>
<td align="char" char=".">0.8888</td>
<td align="char" char=".">0.878</td>
<td align="char" char=".">0.8834</td>
<td align="char" char=".">0.9341</td>
</tr>
<tr>
<td align="left">4e-4</td>
<td align="char" char=".">0.8902</td>
<td align="char" char=".">0.9</td>
<td align="char" char=".">0.878</td>
<td align="char" char=".">0.8888</td>
<td align="char" char=".">0.9388</td>
</tr>
<tr>
<td align="left">5e-4</td>
<td align="char" char=".">0.8841</td>
<td align="char" char=".">0.8705</td>
<td align="char" char=".">0.9024</td>
<td align="char" char=".">0.8862</td>
<td align="char" char=".">0.9375</td>
</tr>
<tr>
<td align="left">6e-4</td>
<td align="char" char=".">0.8902</td>
<td align="char" char=".">0.8902</td>
<td align="char" char=".">0.8902</td>
<td align="char" char=".">0.8902</td>
<td align="char" char=".">0.9298</td>
</tr>
<tr>
<td align="left">7e-4</td>
<td align="char" char=".">0.9085</td>
<td align="char" char=".">0.9135</td>
<td align="char" char=".">0.9024</td>
<td align="char" char=".">0.9079</td>
<td align="char" char=".">0.9301</td>
</tr>
<tr>
<td align="left">8e-4</td>
<td align="char" char=".">0.8902</td>
<td align="char" char=".">0.9102</td>
<td align="char" char=".">0.8658</td>
<td align="char" char=".">0.8875</td>
<td align="char" char=".">0.9144</td>
</tr>
<tr>
<td align="left">9e-4</td>
<td align="char" char=".">0.8841</td>
<td align="char" char=".">0.8987</td>
<td align="char" char=".">0.8658</td>
<td align="char" char=".">0.8819</td>
<td align="char" char=".">0.9207</td>
</tr>
<tr>
<td align="left">1e-3</td>
<td align="char" char=".">0.8658</td>
<td align="char" char=".">0.8571</td>
<td align="char" char=".">0.878</td>
<td align="char" char=".">0.8674</td>
<td align="char" char=".">0.9162</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: The best results are highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Since the CNN kernel represents several amino acids on an anti-cancer peptide sequence sharing the same parameters during the process of convolution. Therefore, the different combination of kernels means that the sequence of the anticancer peptide is affected by different combinations of several amino acids. Modifying the combination of the kernels may improve the effect of the model. Therefore, the combination of the kernel is also a very important parameter. As shown in <xref ref-type="table" rid="T5">Table 5</xref>, when the combination of kernels &#x3d; (<xref ref-type="bibr" rid="B19">Li et al., 2006</xref>; <xref ref-type="bibr" rid="B22">Plumb et al., 2019</xref>; <xref ref-type="bibr" rid="B28">Song et al., 2020</xref>), the model achieved the best effect. Four of the five main evaluation indexes are the best. Accuracy, precision, AUC, and F1-score are the highest. This means that when using multi-kernel CNN to extract features from the model, selecting 1, 3, and 5 amino acid combinations for convolution calculation, and then combining these three features to obtain the best model effect. Our model adopts the combination of one, three, and five amino acids, which is better than considering all the amino acid sequences or just considering the properties of a single amino acid. This is the excellence of the CNN model.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>The performance of the ACP-MCAM model affected by kernel combination.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Kernel</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1-Score</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="char" char=".">0.8536</td>
<td align="char" char=".">0.8372</td>
<td align="char" char=".">0.878</td>
<td align="char" char=".">0.8571</td>
<td align="char" char=".">0.9177</td>
</tr>
<tr>
<td align="left">3</td>
<td align="char" char=".">0.8475</td>
<td align="char" char=".">0.8275</td>
<td align="char" char=".">0.878</td>
<td align="char" char=".">0.852</td>
<td align="char" char=".">0.932</td>
</tr>
<tr>
<td align="left">5</td>
<td align="char" char=".">0.8353</td>
<td align="char" char=".">0.8021</td>
<td align="char" char=".">0.8902</td>
<td align="char" char=".">0.8439</td>
<td align="char" char=".">0.9143</td>
</tr>
<tr>
<td align="left">7</td>
<td align="char" char=".">0.8475</td>
<td align="char" char=".">0.8131</td>
<td align="char" char=".">0.9024</td>
<td align="char" char=".">0.8554</td>
<td align="char" char=".">0.9158</td>
</tr>
<tr>
<td align="left">1 &#x2b; 3</td>
<td align="char" char=".">0.8658</td>
<td align="char" char=".">0.8488</td>
<td align="char" char=".">0.8902</td>
<td align="char" char=".">0.869</td>
<td align="char" char=".">0.9439</td>
</tr>
<tr>
<td align="left">1 &#x2b; 5</td>
<td align="char" char=".">0.8597</td>
<td align="char" char=".">0.8831</td>
<td align="char" char=".">0.8292</td>
<td align="char" char=".">0.8553</td>
<td align="char" char=".">0.9244</td>
</tr>
<tr>
<td align="left">1 &#x2b; 7</td>
<td align="char" char=".">0.8109</td>
<td align="char" char=".">0.8591</td>
<td align="char" char=".">0.7439</td>
<td align="char" char=".">0.7973</td>
<td align="char" char=".">0.8856</td>
</tr>
<tr>
<td align="left">3 &#x2b; 5</td>
<td align="char" char=".">0.8536</td>
<td align="char" char=".">0.8536</td>
<td align="char" char=".">0.8536</td>
<td align="char" char=".">0.8536</td>
<td align="char" char=".">0.9118</td>
</tr>
<tr>
<td align="left">3 &#x2b; 7</td>
<td align="char" char=".">0.7987</td>
<td align="char" char=".">0.7752</td>
<td align="char" char=".">0.8414</td>
<td align="char" char=".">0.807</td>
<td align="char" char=".">0.9015</td>
</tr>
<tr>
<td align="left">5 &#x2b; 7</td>
<td align="char" char=".">0.817</td>
<td align="char" char=".">0.8095</td>
<td align="char" char=".">0.8292</td>
<td align="char" char=".">0.8192</td>
<td align="char" char=".">0.9028</td>
</tr>
<tr>
<td align="left">1 &#x2b; 3&#x2b;5</td>
<td align="char" char=".">
<bold>0.9085</bold>
</td>
<td align="char" char=".">
<bold>0.9589</bold>
</td>
<td align="char" char=".">
<bold>0.8536</bold>
</td>
<td align="char" char=".">
<bold>0.9032</bold>
</td>
<td align="char" char=".">
<bold>0.9479</bold>
</td>
</tr>
<tr>
<td align="left">1 &#x2b; 3&#x2b;7</td>
<td align="char" char=".">0.8597</td>
<td align="char" char=".">0.8831</td>
<td align="char" char=".">0.8292</td>
<td align="char" char=".">0.8553</td>
<td align="char" char=".">0.9074</td>
</tr>
<tr>
<td align="left">1 &#x2b; 5&#x2b;7</td>
<td align="char" char=".">0.8353</td>
<td align="char" char=".">0.8666</td>
<td align="char" char=".">0.7926</td>
<td align="char" char=".">0.828</td>
<td align="char" char=".">0.9131</td>
</tr>
<tr>
<td align="left">3 &#x2b; 5&#x2b;7</td>
<td align="char" char=".">0.8719</td>
<td align="char" char=".">0.8765</td>
<td align="char" char=".">0.8658</td>
<td align="char" char=".">0.8711</td>
<td align="char" char=".">0.9434</td>
</tr>
<tr>
<td align="left">1 &#x2b; 3&#x2b;5 &#x2b; 7</td>
<td align="char" char=".">0.8719</td>
<td align="char" char=".">0.8961</td>
<td align="char" char=".">0.8414</td>
<td align="char" char=".">0.8679</td>
<td align="char" char=".">0.9158</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: The best results are highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-5">
<title>Ablation Experiments</title>
<p>We compared our model to the ACP164 data set for ablation experiments. It can be seen that the experiment is mainly to compare three models: the embedding attention model, CNN attention model, and Multi-kernel CNN attention model. From <xref ref-type="table" rid="T6">Table 6</xref> and <xref ref-type="fig" rid="F3">Figure 3</xref>, we can observe the performance comparison of three different embedding methods. In general, the performance of multi-kernel CNN is better than that of all existing methods on the ACP dataset, indicating that Multi-kernel CNN&#x2019;s embedding method is more powerful than models based on other embedding features.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>The performance of three different models on three different peptide datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Model</th>
<th align="center">Accuracy</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Embedding &#x2b; ACP</td>
<td align="char" char=".">0.8719</td>
<td align="char" char=".">0.9178</td>
<td align="char" char=".">0.817</td>
<td align="char" char=".">0.8645</td>
<td align="char" char=".">0.8719</td>
</tr>
<tr>
<td align="left">Embedding_cnn &#x2b; ACP</td>
<td align="char" char=".">0.8658</td>
<td align="char" char=".">0.8947</td>
<td align="char" char=".">0.8292</td>
<td align="char" char=".">0.8607</td>
<td align="char" char=".">0.9321</td>
</tr>
<tr>
<td align="left">Embedding_multicnn &#x2b; ACP</td>
<td align="char" char=".">0.9085</td>
<td align="char" char=".">0.9589</td>
<td align="char" char=".">0.8536</td>
<td align="char" char=".">0.9032</td>
<td align="char" char=".">0.9479</td>
</tr>
<tr>
<td align="left">Embedding &#x2b; NPs</td>
<td align="char" char=".">0.8343</td>
<td align="char" char=".">0.8197</td>
<td align="char" char=".">0.8571</td>
<td align="char" char=".">0.8380</td>
<td align="char" char=".">0.8840</td>
</tr>
<tr>
<td align="left">Embedding_cnn &#x2b; NPs</td>
<td align="char" char=".">0.8229</td>
<td align="char" char=".">0.8192</td>
<td align="char" char=".">0.8286</td>
<td align="char" char=".">0.8239</td>
<td align="char" char=".">0.8894</td>
</tr>
<tr>
<td align="left">Embedding_multicnn &#x2b; NPs</td>
<td align="char" char=".">0.8400</td>
<td align="char" char=".">0.8479</td>
<td align="char" char=".">0.8285</td>
<td align="char" char=".">0.8381</td>
<td align="char" char=".">0.9063</td>
</tr>
<tr>
<td align="left">Embedding &#x2b; AFPs</td>
<td align="char" char=".">0.8625</td>
<td align="char" char=".">0.8371</td>
<td align="char" char=".">0.9003</td>
<td align="char" char=".">0.8675</td>
<td align="char" char=".">0.9161</td>
</tr>
<tr>
<td align="left">Embedding_cnn &#x2b; AFPs</td>
<td align="char" char=".">0.9038</td>
<td align="char" char=".">0.8803</td>
<td align="char" char=".">0.9347</td>
<td align="char" char=".">0.9067</td>
<td align="char" char=".">0.9580</td>
</tr>
<tr>
<td align="left">Embedding_multicnn &#x2b; AFPs</td>
<td align="char" char=".">0.8762</td>
<td align="char" char=".">0.8119</td>
<td align="char" char=".">0.9793</td>
<td align="char" char=".">0.8878</td>
<td align="char" char=".">0.9677</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: The best results of different dataset are highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Performance comparison of ACP-MCAM and existing methods. The left figure is the ROC curves of different models on the ACP dataset. The right figure is the PR curve of different models on the ACP dataset.</p>
</caption>
<graphic xlink:href="fgene-13-887894-g003.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> is a more intuitive comparison between the several embedding methods of our model, including the ROC curve and precision-recall (PR) curve. From <xref ref-type="fig" rid="F3">Figure 3</xref>, we can see that in both the ROC curve and PR curve, the embedding method of the embedding_multi_cnn can achieve the best effect of the model. Note that the experimental results in this section only reflect the performance of the model on the ACP data set, and it is difficult to avoid certain deviations. Therefore, we evaluate the performance of this model through experiments on other peptide datasets (NPs and AFP). The dataset is shown in <xref ref-type="table" rid="T1">Table 1</xref>. NPs1400 was used as the training set, NPs350 was used as the test set; AFPs2336 was used as the training set, and AFPs582 was used as the test set to verify the model.</p>
<p>The results in <xref ref-type="table" rid="T6">Table 6</xref> show that Multi-kernel CNN performs best on the ACP dataset and AFPs dataset, especially in ACC and AUC. On the NPs data set, CNN has the best feature extraction effect. Therefore, it can be inferred that on the NPs data set, every three consecutive amino acids were regarded as an amino acid group for classification prediction to achieve the best effect.</p>
</sec>
<sec id="s3-6">
<title>Feature Representations and Visualization</title>
<p>Principal Component Analysis (PCA) (<xref ref-type="bibr" rid="B27">Smith, 2002</xref>) is a common linear dimensionality reduction method, while t-distributed Stochastic Neighbor Embedding (TSNE) (<xref ref-type="bibr" rid="B13">Laurens and Hinton, 2008</xref>) is a non-linear dimensionality reduction method. Due to different principles and mechanisms, TSNE runs slower, while PCA is relatively fast. PCA transforms a set of potentially correlated variables into a set of linearly uncorrelated variables through orthogonal transformation, and the transformed set of variables is called principal components. The idea of PCA is to map n-dimensional features to k-dimensions (k &#x3c; n), which are brand new orthogonal features. In this paper, k is equal to 2. The basic idea of TSNE is that similar data points in high-dimensional space map to similar distances in low-dimensional space. The attribute information retained by TSNE is more representative and can relatively reflect the differences between samples. To visually verify the effectiveness of the ACP-MCAM model and improve the interpretability of the model, this paper uses principal component analysis (PCA) and t-distributed stochastic neighborhood embedding (TSNE) to learn the high-dimensionality of ACP sequences at different stages. The high-dimensional feature representation vectors of anticancer peptide sequences at different stages are reduced to a two-dimensional plane for easy visualization, and the results are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Dimension reduction of each samples on ACP500 and ACP164 dataset by TSNE and PCA.</p>
</caption>
<graphic xlink:href="fgene-13-887894-g004.tif"/>
</fig>
<p>The figure shows that in the ACP dataset, the positive examples (represented by train:1 and purple dots) and negative examples (represented by train:0 and blue dots) in the training dataset are mixed in the initial stage because they are initialized randomly. The same is in the test dataset (positive examples are represented by test:0 and red dots, and negative examples are represented by test:1 and blue dots), which indicates that the model has no distinguishing ability before training. As the training epoch number increases, positive and negative samples are gradually separated from the sample points. We can observe that in the training dataset and the test dataset, the embedding vectors of the ACP samples almost belong to the same cluster, and after training, the positive and negative examples have similar distributions, which indicates that the model has indeed learned feature of the positive and negative samples. This shows that the model in this paper can learn the common features and distinguishing features of positive and negative cases.</p>
<p>In addition, there are many ACPs in the negative cluster, but few non-ACPs in the positive cluster, which explains the reason why the performance of SP is better than SE to some extent. We speculate that those ACPs predicted to be negative samples have characteristics that our method cannot capture. Therefore, the unique physical and chemical properties of these indistinguishable samples should be further studied in the future.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>Conclusion</title>
<p>A very important point in deep learning is how to extract features from data. The quality of the extracted features will largely affect the effectiveness of the model. The advantage of natural language processing (NLP) is that it can effectively extract word embedding and sequence information from sequence data, and use it for subsequent specific tasks. Our method can automatically learn useful information from the amino acid sequence data of anti-cancer peptides and perform feature representation on node features and sequence features.</p>
<p>In this work, we proposed a new predictive model called ACP-MCAM. This is a powerful bioinformatics tool. The model can predict anti-cancer peptides based on a convolutional neural network and self-attention mechanism network, which can extract effective amino acid nodes and anti-cancer peptide sequence information. The advantage of ACP-MCAM is that it can effectively use the position information and the information of the amino acid node cluster. The ACP-MCAM model mainly includes the following modules: embedding layer, multi-kernel convolutional neural network layer, position coding layer, attention encoding layer, and task output layer. The experimental results of 10-fold cross-validation and independent testing show that this predictor can effectively distinguish anti-cancer peptides from non-anti-cancer peptides. Moreover, we used the model to predict neuropeptides and antifungal peptides and achieved good prediction results. The excellent predictive ability of this model will accelerate its application in cancer treatment.</p>
<p>Our model has achieved good prediction performance, but there are still some shortcomings to be overcome. First of all, the prediction performance of the model fluctuates greatly, and different parameters have a greater impact on the prediction results. The main reason is that 500 pieces of data in the training set and 164 pieces of data in the test set are too small for deep learning to train all parameters. This is where we will strive to improve in the future. In future work, we will expand more datasets and try more computing techniques, such as pre-training strategies for automatic feature extraction, to achieve more accurate and better predictions. Second, in our current case study, we only made computer model predictions based on the original database and did not verify it in silicon experiments. In the future, we will plan to cooperate with biologists to conduct wet laboratory experiments to verify the predicted results.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>FL and WZ initialized the research project and designed the experiments; XW and PX designed software, wrote the codes, performed the experiments, and wrote the paper. XL participated in performing the experiments and wrote the paper. All authors reviewed the manuscript and agree to be accountable for the content of the work.</p>
</sec>
<sec sec-type="COI-statement" id="s7">
<title>Conflict of Interest</title>
<p>PX was employed by the company Chongqing Michong Technology Co., Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s8">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>Thanks to all the peer reviewers for their opinions and suggestions.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Agrawal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bhagat</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mahalwal</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Raghava</surname>
<given-names>G. P. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>AntiCP 2.0: An Updated Model for Predicting Anticancer Peptides</article-title>. <source>Brief. Bioinform</source>. <volume>22</volume>. <fpage>bbaa153</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa153</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Atul</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Abhishek</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Priya</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sudheer</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Minakshi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Deepika</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>CancerPPD: a Database of Anticancer Peptides and Proteins</article-title>. <source>Nucleic Acids Res.</source> <volume>43</volume> (<issue>D1</issue>), <fpage>837</fpage>&#x2013;<lpage>843</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gku892</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Balachandran</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shaherin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Tae</surname>
<given-names>H. S.</given-names>
</name>
<name>
<surname>Leyi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gwang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>mAHTPred: a Sequence-Based Meta-Predictor for Improving the Prediction of Anti-hypertensive Peptides Using Effective Feature Representation</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>2757</fpage>&#x2013;<lpage>2765</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty1047</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barras</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Widmann</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Promises of Apoptosis-Inducing Peptides in Cancer Therapeutics</article-title>. <source>Cpb</source> <volume>12</volume>, <fpage>1153</fpage>&#x2013;<lpage>1165</lpage>. <pub-id pub-id-type="doi">10.2174/138920111796117337</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>iACP: a Sequence-Based Tool for Identifying Anticancer Peptides</article-title>. <source>Oncotarget</source> <volume>7</volume> (<issue>13</issue>), <fpage>16895</fpage>&#x2013;<lpage>16909</lpage>. <pub-id pub-id-type="doi">10.18632/oncotarget.7815</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clamp</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jayson</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>The Clinical Development of the Bryostatins</article-title>. <source>Anti-Cancer Drugs</source> <volume>13</volume> (<issue>7</issue>), <fpage>673</fpage>&#x2013;<lpage>683</lpage>. <pub-id pub-id-type="doi">10.1097/00001813-200208000-00001</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Farabet</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Couprie</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Najman</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Learning Hierarchical Features for Scene Labeling</article-title>. <source>IEEE Trans. Pattern Anal. Machine Intelligence</source> <volume>35</volume> (<issue>8</issue>), <fpage>1915</fpage>&#x2013;<lpage>1929</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2012.231</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gaspar</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Veiga</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Castanho</surname>
<given-names>M. A. R. B.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>From Antimicrobial to Anticancer Peptides. A Review</article-title>. <source>Front. Microbiol.</source> <volume>4</volume>, <fpage>294</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2013.00294</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gutierrez</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Glanzner</surname>
<given-names>W. G.</given-names>
</name>
<name>
<surname>Chemeris</surname>
<given-names>R. O.</given-names>
</name>
<name>
<surname>Rigo</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Comim</surname>
<given-names>F. V.</given-names>
</name>
<name>
<surname>Bordignon</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Gonadotoxic Effects of Busulfan in Two Strains of Mice</article-title>. <source>Reprod. Toxicol.</source> <volume>59</volume> (<issue>9</issue>), <fpage>31</fpage>&#x2013;<lpage>39</lpage>. <pub-id pub-id-type="doi">10.1016/j.reprotox.2015.09.002</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hajisharifi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Piryaiee</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mohammad Beigi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Behbahani</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mohabatkar</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Predicting Anticancer Peptides with Chou&#x2032;s Pseudo Amino Acid Composition and Investigating Their Mutagenicity via Ames Test</article-title>. <source>J. Theor. Biol.</source> <volume>341</volume>, <fpage>34</fpage>&#x2013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2013.08.037</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Otvos</surname>
<given-names>L</given-names>
</name>
</person-group> (<year>2008</year>). <source>Peptide-based Drug Design: Here and Now</source>, <publisher-name>Methods in Molecular Biology</publisher-name> <volume>494</volume>. <fpage>1</fpage>. <pub-id pub-id-type="doi">10.1007/978-1-59745-419-3_1</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kamisli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ciftci</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Kaya</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Cetin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kamisli</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Ozcan</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Hesperidin Protects Brain and Sciatic Nerve Tissues against Cisplatin-Induced Oxidative, Histological and Electromyographical Side Effects in Rats</article-title>. <source>Toxicol. Ind. Health</source> <volume>31</volume> (<issue>9</issue>), <fpage>841</fpage>&#x2013;<lpage>851</lpage>. <pub-id pub-id-type="doi">10.1177/0748233713483192</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laurens</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Visualizing Data Using T-SNE</article-title>. <source>J. Machine Learn. Res.</source> <volume>9</volume> (<issue>2605</issue>), <fpage>2579</fpage>&#x2013;<lpage>2605</lpage>. </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lecun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bottou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Haffner</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Gradient-based Learning Applied to Document Recognition</article-title>. <source>Proc. IEEE</source> <volume>86</volume> (<issue>11</issue>), <fpage>2278</fpage>&#x2013;<lpage>2324</lpage>. <pub-id pub-id-type="doi">10.1109/5.726791</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leyi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Huangrong</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jiangning</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ran</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>ACPred-FL: a Sequence-Based Predictor Using Effective Feature Representation to Improve the Prediction of Anti-cancer Peptides</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>23</issue>), <fpage>4007</fpage>&#x2013;<lpage>4016</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty451</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leyi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ran</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Quan</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>PEPred-Suite: Improved and Robust Prediction of Therapeutic Peptides Using Adaptive Feature Representation Learning</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>21</issue>), <fpage>4272</fpage>&#x2013;<lpage>4280</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz246</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leyi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wenjia</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Adeel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ran</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lizhen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Balachandran</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Computational Prediction and Interpretation of Cell-specific Replication Origin Sites from Multiple Eukaryotes by Exploiting Stacking Framework</article-title>. <source>Brief. Bioinform.</source> <volume>22</volume>, <fpage>275</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa275</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>PSBinder: A Web Service for Predicting Polystyrene Surface-Binding Peptides</article-title>. <source>Biomed. Res. Int.</source> <volume>2017</volume> (<issue>6</issue>), <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1155/2017/5761517</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z. R.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H. H.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>L. Y.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y. Z.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>PROFEAT: a Web Server for Computing Structural and Physicochemical Features of Proteins and Peptides from Amino Acid Sequence</article-title>. <source>Nucleic Acids Res.</source> <volume>34</volume> (<issue>Suppl. l_2</issue>), <fpage>W32</fpage>&#x2013;<lpage>W37</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkl305</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mehedi</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Nalini</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shaherin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gwang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Watshara</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Balachandran</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Improved and Robust Prediction of Hemolytic Peptide and its Activity by Fusing Multiple Feature Representation</article-title>. <source>Bioinformatics</source> <volume>11</volume> (<issue>11</issue>), <fpage>3350</fpage>&#x2013;<lpage>3356</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa160</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Novin</surname>
<given-names>MG</given-names>
</name>
<name>
<surname>Sciences</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>The Effect of Busulfan on Body Weight, Testis Weight and MDA Enzymes in Male Rats</article-title>. <source>Int. J. Womens Health Reprod. Sci.</source> <volume>2</volume> (<issue>5</issue>), <fpage>316</fpage>&#x2013;<lpage>319</lpage>. <pub-id pub-id-type="doi">10.15296/ijwhr.2014.52</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Plumb</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Balaji</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Rabbab</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Natividad</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Yoshiyuki</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sathiyamoorthy</surname>
<given-names>V. N.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Cucurbit[7]uril Encapsulated Cisplatin Overcomes Cisplatin Resistance via a Pharmacokinetic Effect</article-title>. <source>Metallomics</source> <volume>35</volume> (<issue>8</issue>), <fpage>1391</fpage>&#x2013;<lpage>1400</lpage>. <pub-id pub-id-type="doi">10.1039/c2mt20054f</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Progress in the Mechanisms of Anticancer Peptides</article-title>. <source>Sheng Wu Gong Cheng Xue Bao</source> <volume>35</volume> (<issue>8</issue>), <fpage>1391</fpage>&#x2013;<lpage>1400</lpage>. <pub-id pub-id-type="doi">10.13345/j.cjb.190033</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>ACPred-Fuse: Fusing Multi-View Information Improves the Prediction of Anticancer Peptides</article-title>. <source>Brief Bioinform</source> <volume>21</volume> (<issue>5</issue>), <fpage>1846</fpage>&#x2013;<lpage>1855</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz088</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ratain</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Geary</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Undevia</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Coronado</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Alfaro</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Iglesias</surname>
<given-names>J. L.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>First-in-human, Phase I Study of Elisidepsin (PM02734) Administered as a 30-min or as a 3-hour Intravenous Infusion Every Three Weeks in Patients with Advanced Solid Tumors</article-title>. <source>Invest. New Drugs</source> <volume>33</volume>, <fpage>901</fpage>&#x2013;<lpage>910</lpage>. <pub-id pub-id-type="doi">10.1007/s10637-015-0247-1</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ryu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yeom</surname>
<given-names>J.-H.</given-names>
</name>
<name>
<surname>Joo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Rediscovery of Antimicrobial Peptides as Therapeutic Agents</article-title>. <source>J. Microbiol.</source> <volume>59</volume> (<issue>2</issue>), <fpage>113</fpage>&#x2013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1007/s12275-021-0649-z</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smith</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>A Tutorial on Principal Components Analysis</article-title>. <source>Inf. Fusion</source> <volume>51</volume>, <fpage>52</fpage>. </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhuang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Min</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Comprehensive Review and Comparison for Anticancer Peptides Identification Models</article-title>. <source>Curr. Protein Pept. Sci.</source> <volume>21</volume>, <fpage>223</fpage>&#x2013;<lpage>231</lpage>. <pub-id pub-id-type="doi">10.2174/1389203721666200117162958</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Manavalan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Empirical Comparison and Analysis of Web-Based Cell-Penetrating Peptide Prediction Tools</article-title>. <source>Brief. Bioinform.</source> <volume>21</volume> (<issue>2</issue>), <fpage>408</fpage>&#x2013;<lpage>420</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bby124</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep-Resp-Forest: A Deep forest Model to Predict Anti-cancer Drug Response</article-title>. <source>Methods</source> <volume>166</volume> (<issue>7</issue>), <fpage>91</fpage>&#x2013;<lpage>102</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymeth.2019.02.009</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tyagi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kapoor</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chaudhary</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Gautam</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Raghava</surname>
<given-names>G. P. S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>In Silico models for Designing and Discovering Novel Anticancer Peptides</article-title>. <source>Sci. Rep.</source> <volume>3</volume>, <fpage>2984</fpage>&#x2013;<lpage>2996</lpage>. <pub-id pub-id-type="doi">10.1038/srep02984</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Acker</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Van Malderen</surname>
<given-names>S. J. M.</given-names>
</name>
<name>
<surname>Van Heerden</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mcduffie</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Cuyckens</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Vanhaecke</surname>
<given-names>F. F. J. A. C. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>High-resolution Laser Ablation-Inductively Coupled Plasma-Mass Spectrometry Imaging of Cisplatin-Induced Nephrotoxic Side Effects</article-title>. <source>Analytica Chim. Acta</source> <volume>945</volume> (<issue>9</issue>), <fpage>23</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1016/j.aca.2016.10.014</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Attention Is All You Need</article-title> in <conf-name>Proceedings of the 31st Conference on Neural Information Processing Systems(NIPS 2017)</conf-name> (<conf-loc>USA: Long Beach, CA</conf-loc>), <conf-date>December 4, 2017</conf-date>. </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vijayakumar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ptv</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>ACPP: A Web Server for Prediction and Design of Anti-cancer Peptides</article-title>. <source>Int. J. Pept. Res. Ther.</source> <volume>21</volume> (<issue>1</issue>), <fpage>99</fpage>&#x2013;<lpage>106</lpage>. <pub-id pub-id-type="doi">10.1007/s10989-014-9435-7</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Gou</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Rehg</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F.-Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Parallel Vision for Perception and Understanding of Complex Scenes: Methods, Framework, and Perspectives</article-title>. <source>Artif. Intell. Rev.</source> <volume>48</volume> (<issue>3</issue>), <fpage>299</fpage>&#x2013;<lpage>329</lpage>. <pub-id pub-id-type="doi">10.1007/s10462-017-9569-z</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Comparative Analysis and Prediction of Quorum-sensing Peptides Using Feature Representation Learning and Machine Learning Algorithms</article-title>. <source>Brief. Bioinform.</source> <volume>21</volume> (<issue>1</issue>), <fpage>106</fpage>&#x2013;<lpage>119</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bby107</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Z. S.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>CPPred-RF: A Sequence-Based Predictor for Identifying Cell-Penetrating Peptides and Their Uptake Efficiency</article-title>. <source>J. Proteome Res.</source> <volume>16</volume> (<issue>5</issue>), <fpage>2044</fpage>&#x2013;<lpage>2053</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jproteome.7b00019</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ying</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Beifang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ying</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Limin</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Weizhong</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>A Web Server for Clustering and Comparing Biological Sequences</article-title>. <source>Bioinformatics</source> <volume>26</volume>, <fpage>680</fpage>&#x2013;<lpage>682</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq003</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Convolutional Neural Networks for Hyperspectral Image Classification</article-title>. <source>Neurocomputing</source> <volume>219</volume>, <fpage>88</fpage>&#x2013;<lpage>98</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2016.09.010</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>