<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">845747</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.845747</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>BBPpredict: A Web Service for Identifying Blood-Brain Barrier Penetrating Peptides</article-title>
<alt-title alt-title-type="left-running-head">Chen et al.</alt-title>
<alt-title alt-title-type="right-running-head">BBB Penetrating Peptides Predictor</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Xue</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1617241/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Qianyue</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1778044/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Bowen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1778181/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lu</surname>
<given-names>Chunying</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1778176/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Shanshan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1778048/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Long</surname>
<given-names>Jinjin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1778116/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>He</surname>
<given-names>Bifang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/877995/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chen</surname>
<given-names>Heng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/998722/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Huang</surname>
<given-names>Jian</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/449350/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Medical College</institution>, <institution>Guizhou University</institution>, <addr-line>Guiyang</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Life Science and Technology</institution>, <institution>University of Electronic Science and Technology of China</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1385377/overview">Chuan Dong</ext-link>, Wuhan University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1064518/overview">Leyi Wei</ext-link>, Shandong University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/273907/overview">Zunnan Huang</ext-link>, Guangdong Medical University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Bifang He, <email>bfhe@gzu.edu.cn</email>; Heng Chen, <email>hchen13@gzu.edu.cn</email>; Jian Huang, <email>hj@uestc.edu.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>845747</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>12</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Chen, Zhang, Li, Lu, Yang, Long, He, Chen and Huang.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Chen, Zhang, Li, Lu, Yang, Long, He, Chen and Huang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Blood-brain barrier (BBB) is a major barrier to drug delivery into the brain in the treatment of central nervous system (CNS) diseases. Blood-brain barrier penetrating peptides (BBPs), a class of peptides that can cross BBB through various mechanisms without damaging BBB, are effective drug candidates for CNS diseases. However, identification of BBPs by experimental methods is time-consuming and laborious. To discover more BBPs as drugs for CNS disease, it is urgent to develop computational methods that can quickly and accurately identify BBPs and non-BBPs. In the present study, we created a training dataset that consists of 326 BBPs derived from previous databases and published manuscripts and 326 non-BBPs collected from UniProt, to construct a BBP predictor based on sequence information. We also constructed an independent testing dataset with 99 BBPs and 99 non-BBPs. Multiple machine learning methods were compared based on the training dataset via a nested cross-validation. The final BBP predictor was constructed based on the training dataset and the results showed that random forest (RF) method outperformed other classification algorithms on the training and independent testing dataset. Compared with previous BBP prediction tools, the RF-based predictor, named BBPpredict, performs considerably better than state-of-the-art BBP predictors. BBPpredict is expected to contribute to the discovery of novel BBPs, or at least can be a useful complement to the existing methods in this area. BBPpredict is freely available at <ext-link ext-link-type="uri" xlink:href="http://i.uestc.edu.cn/BBPpredict/cgi-bin/BBPpredict.pl">http://i.uestc.edu.cn/BBPpredict/cgi-bin/BBPpredict.pl</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>blood-brain barrier</kwd>
<kwd>random forest (RF)</kwd>
<kwd>nested cross-validation</kwd>
<kwd>computational method</kwd>
<kwd>blood-brain barrier penetrating peptides (BBPs)</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Blood-brain barrier (BBB) highly protects the central nervous system (CNS) (<xref ref-type="bibr" rid="B31">Nance et al., 2022</xref>), preventing 98% of small molecules and 100% of large molecules from entering the brain (<xref ref-type="bibr" rid="B34">S&#xe1;nchez-Navarro et al., 2017</xref>). It is the main obstacle for drug delivery into the brain (<xref ref-type="bibr" rid="B1">Banks, 2016</xref>). Therefore, exploring methods for drugs to penetrate BBB is a research hotpot in the development of drugs for CNS disorders (<xref ref-type="bibr" rid="B36">Terstappen et al., 2021</xref>).</p>
<p>Blood-brain barrier penetrating peptides (BBPs) can cross the BBB through various mechanisms without destroying the integrity of BBB (<xref ref-type="bibr" rid="B37">Van Dorpe et al., 2012</xref>; <xref ref-type="bibr" rid="B33">Oller-Salvia et al., 2016</xref>). It has been reported that partial BBPs can transfer drugs into the brain, which provides a new avenue for the development of drugs for CNS diseases (<xref ref-type="bibr" rid="B44">Zhou et al., 2021</xref>). Furthermore, because of their characteristics of easy synthesis, satisfactory effect, low toxicity and wide selectivity (<xref ref-type="bibr" rid="B30">Muttenthaler et al., 2021</xref>), BBPs show broad application prospects as carriers or therapeutic agents for CSN diseases treatment (<xref ref-type="bibr" rid="B44">Zhou et al., 2021</xref>). Nonaka et al. reported that IF7, an annexin A1-binding peptide, could overcome BBB and deliver chemotherapeutics to target brain tumors (<xref ref-type="bibr" rid="B32">Nonaka et al., 2020</xref>). Xie and coworkers demonstrated that d-peptide ligand of angiopep-2 modified nanoprobes could cross BBB and locate glioma sites (<xref ref-type="bibr" rid="B42">Xie et al., 2021</xref>). Lim and collaborators found that dNP2 peptide could penetrate BBB and deliver ctCTLA-4 protein to ameliorate autoimmune encephalomyelitis in mouse models (<xref ref-type="bibr" rid="B29">Lim et al., 2015</xref>). Kurzrock and Drappatz et al. showed that ANG1005 or GRN1005, a conjugate of angiopep-2 and paclitaxel, has reached clinical study for the treatment of glioma (<xref ref-type="bibr" rid="B26">Kurzrock et al., 2012</xref>; <xref ref-type="bibr" rid="B15">Drappatz et al., 2013</xref>).</p>
<p>There have been two BBP databases published to date, Brainpeps (<xref ref-type="bibr" rid="B37">Van Dorpe et al., 2012</xref>) and B3Pdb (<xref ref-type="bibr" rid="B25">Kumar et al., 2021b</xref>), since BBPs became candidates for developing peptide agents for managing CNS disorders. These studies are undoubtedly a strong boost to the development of medications for CNS diseases. However, the discovery of BBPs by wet-lab experiment is time-consuming and complex, and only hundreds of BBPs have been identified experimentally to date. Construction of computational methods for the identification of BBPs is very valuable for developing therapeutics for CSN diseases. Machine learning methods have been successfully applied to the classification of various peptides, such as cell-penetrating peptides (<xref ref-type="bibr" rid="B40">Wei et al., 2017a</xref>; <xref ref-type="bibr" rid="B41">Wei et al., 2017b</xref>; <xref ref-type="bibr" rid="B23">Kumar et al., 2018</xref>), antimicrobial peptides (<xref ref-type="bibr" rid="B3">Bhadra et al., 2018</xref>), anticancer peptides (<xref ref-type="bibr" rid="B28">Li and Wang, 2016</xref>). There are also two BBP predictors, BBPpred (<xref ref-type="bibr" rid="B14">Dai et al., 2021</xref>) and B3Pred (<xref ref-type="bibr" rid="B24">Kumar et al., 2021a</xref>), have published successively for identifying BBPs. BBPpred is based on logistic regression to identify BBPs, while B3Pred uses random forest (RF) to predict BBPs. Considering the low sample complexity of these two classifiers, the performance of computational models for identifying BBPs can be improved.</p>
<p>In this work, we collected more BBPs from existing databases (<xref ref-type="bibr" rid="B37">Van Dorpe et al., 2012</xref>; <xref ref-type="bibr" rid="B25">Kumar et al., 2021b</xref>) and published literatures to construct a new BBP predictor named BBPpredict, which is an online web service and freely available at <ext-link ext-link-type="uri" xlink:href="http://i.uestc.edu.cn/BBPpredict/cgi-bin/BBPpredict.pl">http://i.uestc.edu.cn/BBPpredict/cgi-bin/BBPpredict.pl</ext-link>. By comparing the results of the nested five-fold cross-validation and independent testing dataset of various machine learning predictors, the RF-based model showed the best prediction performance. Thus, BBPpredict was implemented by using RF. We expect BBPpredict will help researchers find more novel BBPs.</p>
</sec>
<sec id="s2">
<title>2 Materials and Methods</title>
<sec id="s2-1">
<title>2.1 Datasets</title>
<p>In this work, we selected experimentally validated BBPs as candidate positive samples that were collected from Brainpeps (<xref ref-type="bibr" rid="B37">Van Dorpe et al., 2012</xref>), B3Pdb(<xref ref-type="bibr" rid="B25">Kumar et al., 2021b</xref>), public datasets of BBPpred (<xref ref-type="bibr" rid="B14">Dai et al., 2021</xref>) and B3Pred (<xref ref-type="bibr" rid="B24">Kumar et al., 2021a</xref>), and other published literatures from PubMed with query &#x201c;(((Brain [Title/Abstract]) OR (blood&#x2013;brain barrier [Title/Abstract])) AND peptide [Title/Abstract]) AND (transport [Title/Abstract] OR transfer [Title/Abstract] OR permeation [Title/Abstract] OR permeability [Title/Abstract])&#x201d;, covering the period 2011&#x2013;2021. BBPs were then preprocessed as follows: 1) the repetitive sequences were eliminated; 2) peptide sequences with ambiguous residues (&#x201c;X&#x201d;, &#x201c;B&#x201d; and &#x201c;Z&#x201d;, etc.) were deleted (<xref ref-type="bibr" rid="B20">He et al., 2016</xref>). Finally, 425 BBPs were remained as positive samples. We also collected 1,304 non-BBPs that were obtained by the following three steps: 1) collect initial sequences from UniProt with the query &#x201c;peptides length: [5 TO 50] NOT blood brain barrier NOT brain NOT brainpeps NOT b3pdb NOT permeation NOT permeability NOT venom NOT toxin NOT transmembrane NOT transport NOT transfer NOT membrane NOT neuro NOT hemolysis AND reviewed: yes&#x201d; (<xref ref-type="bibr" rid="B14">Dai et al., 2021</xref>), 2) remove redundant sequences by using CD-HIT (sequence identity cut-off of 10%) (<xref ref-type="bibr" rid="B14">Dai et al., 2021</xref>), 3) exclude the peptide sequences with ambiguous residues (&#x201c;X&#x201d;, &#x201c;B,&#x201d; and &#x201c;Z&#x201d;, etc.).</p>
</sec>
<sec id="s2-2">
<title>2.2 Training and Independent Testing Datasets</title>
<p>To evaluate the performance of our predictor and existing predictors (BBPpred and B3Pred), 99 BBPs that collected through published literatures and 99 non-BBPs randomly selected from candidate negative samples construct an independent testing dataset that was completely independent of the training dataset of the three predictor models (BBPpred, B3Pred and our proposed BBPpredict) (<xref ref-type="table" rid="T1">Table 1</xref>). The remaining 326 BBPs were used as the positive training dataset. To balance the sample size for training, we randomly selected 326 non-BBPs as the negative training dataset (<xref ref-type="table" rid="T1">Table 1</xref>), whose length distribution is the same as the positive training dataset. All datasets are available for download from <ext-link ext-link-type="uri" xlink:href="http://i.uestc.edu.cn/BBPpredict/download.html">http://i.uestc.edu.cn/BBPpredict/download.html</ext-link>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>List of training dataset and independent testing dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Dataset</th>
<th align="center">Number of BBPs</th>
<th align="center">Number of Non-BBPs</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Training dataset</td>
<td align="char" char=".">326</td>
<td align="char" char=".">326</td>
</tr>
<tr>
<td align="left">Independent testing dataset</td>
<td align="char" char=".">99</td>
<td align="char" char=".">99</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-3">
<title>2.3 Feature Extraction</title>
<p>Feature extraction refers to the transformation of peptide sequences into fixed-length feature vectors, which is an indispensable step for the construction of predictors. In this study, we selected five feature encoding methods, including amino acid composition (AAC), dipeptide composition (DPC), composition of <italic>k</italic>-spaced amino acid group pairs (CKSAAGP, <italic>k</italic> &#x3d; 3), pseudo-amino acid composition (PAAC) and grouped amino acid composition (GAAC) to extract the characteristics of peptide sequence. Here we set the length of a peptide to be <italic>N</italic>, and all feature extraction methods are based on 20 natural amino acids (i.e., &#x201c;ACDEFGHIKLMNPQRSTVWY&#x201d;). Feature extraction was implemented by an in-house script.</p>
<sec id="s2-3-1">
<title>2.3.1 Amino Acid Composition</title>
<p>AAC calculates the frequency of each amino acid in the peptide sequence (<xref ref-type="bibr" rid="B4">Bhasin and Raghava, 2004</xref>). It can be calculated as:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>...</mml:mn>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>N(i)</italic> is the number of the amino acid type <italic>i</italic>.</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Dipeptide Composition</title>
<p>DPC gives 400 descriptors (i.e.&#x201c;<inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x201d;) (<xref ref-type="bibr" rid="B35">Saravanan and Gautham, 2015</xref>). It is defined as:<disp-formula id="e2">
<mml:math id="m3">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>...</mml:mn>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <italic>Nrs</italic> is the number of the dipeptide consisting of amino acids <italic>r</italic> and <italic>s</italic> in the peptide sequence.</p>
</sec>
<sec id="s2-3-3">
<title>2.3.3 Grouped Amino Acid Composition</title>
<p>For the GAAC encoding, 20 natural amino acids are firstly divided into five categories according to their physicochemical properties: amino acid groups g1 (GAVLMI), g2 (FYW), g3 (KRH), g4 (DE) and g5 (STCPNQ). Group g1 belongs to the aliphatic group, g2 aromatic group, g3 positive charge group, g4 negative charged group and g5 uncharged group, respectively. GAAC represents the frequency of each amino acid group (<xref ref-type="bibr" rid="B27">Lee et al., 2011</xref>) and can be described as:<disp-formula id="e3">
<mml:math id="m4">
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>g</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
<mml:mn>4</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
<mml:mn>5</mml:mn>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
<mml:mn>4</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
<mml:mn>5</mml:mn>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf2">
<mml:math id="m5">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the number of amino acids in group g, <italic>N(i)</italic> is the number of the amino acid type <italic>i</italic>.</p>
</sec>
<sec id="s2-3-4">
<title>2.3.4 Composition of <italic>K</italic>-Spaced Amino Acid Group Pairs</title>
<p>CKSAAGP is based on CKSAAP (<xref ref-type="bibr" rid="B9">Chen et al., 2007a</xref>; <xref ref-type="bibr" rid="B7">Chen et al., 2007b</xref>, <xref ref-type="bibr" rid="B8">2008</xref>; <xref ref-type="bibr" rid="B6">Chen et al., 2009</xref>) descriptor and GAAC descriptor, which calculates the frequency of <italic>k</italic>-spaced group pairs. And the detailed calculation of CKSAAGP can refer to (<xref ref-type="bibr" rid="B10">Chen et al., 2018</xref>). In this study, we set k as three by default. And when k &#x3d; 0, CKSAAGP can be calculated as:<disp-formula id="e4">
<mml:math id="m6">
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">total</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">total</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">total</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mn>...</mml:mn>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">total</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mn>25</mml:mn>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>Where <inline-formula id="inf3">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> describes <italic>N-</italic>1<italic>,</italic> <inline-formula id="inf4">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of 0-spaced group pairs.</p>
</sec>
<sec id="s2-3-5">
<title>2.3.5 Pseudo-Amino Acid Composition</title>
<p>PAAC describes the information of two residues order and properties in the peptide sequence. The computation of PAAC is available in (<xref ref-type="bibr" rid="B12">Chou, 2001</xref>; <xref ref-type="bibr" rid="B13">2005</xref>).</p>
<p>After feature extraction, each peptide was encoded by a 550-dimensional feature vector, which was generated by concatenating five types of feature vector.</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Feature Scoring and Selection</title>
<p>Generally, not all features make contribution to the model construction. Partial features make remarkable contributions, while some others make slight contributions (<xref ref-type="bibr" rid="B19">He et al., 2019</xref>). Therefore, feature selection is a very vital step for accomplishing a classifier model with promising classification performance (<xref ref-type="bibr" rid="B43">Zhao et al., 2016</xref>). In this study, F-score method was employed to estimate each feature&#x2019;s contribution. The feature with a greater F-score implies its larger contribution for prediction model. We conducted the following procedures to select more informative features from the 550 features that were extracted from the training dataset. In the first stage, we evaluated the five-fold cross-validation performance of top 92, 184, 275, 367, 458, 550 features for various classification algorithms. In the five-fold cross-validation, the training dataset was equally divided into five subsets, among these five subsets, a subset was used as the testing-set and the other four subsets as the training-set. The division of top 92, 184, 275, 367, 458, 550 features based on the training-set was determined by making (count_max-count_min)/6 as the cut-off point of feature division, where &#x201c;count_max&#x201d; represents the maximum dimension of feature (550 features), and &#x201c;count_min&#x201d; is the minimum dimension of feature (1 feature). In the second stage, according to the five-fold cross-validation results of different classification algorithms, we obtained the number of features n with the highest accuracy. In the third stage, we selected top n features from the 550 features extracted from the training dataset and ranked by F-score in descending order to construct the final model.</p>
</sec>
<sec id="s2-5">
<title>2.5 Classification Model Construction</title>
<p>Eight traditional machine learning algorithms, including decision tree (DT), RF, k-nearest neighbors (KNN), adaptive boosting (AdaBoost), gentle adaptive boosting (GentleBoost), adaptive logistic regression (LogitBoost), linear support vector machine (linearSVM) and radial basis function (RBF) kernel SVM (rbfSVM) were used to build the predictive models based on the features selected by feature selection (see in <xref ref-type="sec" rid="s10">Supplementary Table S3</xref>), respectively. LIBSVM 3.24 (<ext-link ext-link-type="uri" xlink:href="http://www.csie.ntu.edu.tw/%7Ecjlin/libsvm/">http://www.csie.ntu.edu.tw/&#x223c;cjlin/libsvm/</ext-link>) was utilized to accomplish linearSVM and rbfSVM (<xref ref-type="bibr" rid="B5">Chang and Lin, 2011</xref>). DT, RF, KNN, AdaBoost, GentleBoost and LogitBoost are respectively implemented by MATLAB R2021a built-in functions fitcTree, TreeBagger, fitcknn and fitcEnmbles. To compare with deep learning method, a long-short term memory (LSTM) network that realized based on Keras 2.3.1 (tensorflow 2.1.0 as backend) package of python 3.6 was also utilized to construct the classification model (<xref ref-type="bibr" rid="B21">Hochreiter and Schmidhuber, 1997</xref>). The LSTM classification model consisted of one LSTM layer with eight hidden neurons. The non-linear activation function hyperbolic tangent (tanh) was applied to LSTM layer. It should be noted that for LSTM, the vectored sequence of peptide was utilized as classification features and no feature selection was applied. The pseudo code for final model construction can be found in the <xref ref-type="sec" rid="s10">Supplementary Material</xref>.</p>
</sec>
<sec id="s2-6">
<title>2.6 Prediction Assessment</title>
<p>Five evaluation indexes, including accuracy (ACC), sensitivity (SN), specificity (SP), Matthews correlation coefficient (MCC) and the area under the receiver operating characteristic (ROC) curve (AUC), were utilized to quantify the performance of each predictive model. The first four indicators are calculated as follows:<disp-formula id="e5">
<mml:math id="m9">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m10">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m11">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m12">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="normal">&#xd7;</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="normal">&#xd7;</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <italic>TP</italic> describes the number of genuine BBPs which are predicted as BBPs. <italic>FN</italic> represents the number of genuine BBPs that are identified as non-BBPs. Denote <italic>TN</italic> as the number of true non-BBPs classified as non-BBPs and <italic>FP</italic> the number of true non-BBPs identified as BBPs. <italic>SN</italic> and <italic>SP</italic> primarily assess the ability of a predictive model to identify positive and negative samples respectively, while <italic>ACC</italic> and <italic>MCC</italic> investigate the comprehensive capacity of a prediction model to classify both positive and negative samples (<xref ref-type="bibr" rid="B39">Wang et al., 2019</xref>). The AUC score is often utilized to judge the merits and demerits of classifiers. In this study, we selected the optimal predictive model according to the AUC value. The model construction and evaluation were performed at a computational server (Sugon I840-G20, Dawning Information Industry Co., LTD., Beijing, China).</p>
</sec>
<sec id="s2-7">
<title>2.7 Reproducible Analysis</title>
<p>Data analysis reproducibility plays a vital role for achieving an independent verification of the analysis results (<xref ref-type="bibr" rid="B38">Walzer and Vizca&#xed;no, 2020</xref>). In this work, we constructed 100 testing datasets and corresponding training datasets to verify the robustness of the construction method of the BBP predictor. To avoid high similarity between the independent testing dataset and the testing dataset of the reproducible analysis, here each testing dataset consisted of 50 BBPs randomly selected from candidate positive samples (114 BBPs) that are independent of the training datasets of BBPpred and B3Pred and 50 non-BBPs with the same selection rules with BBPs. The model building process based on 100 reconstructed datasets for different classification algorithms (RF, rbfSVM, linearSVM, etc.) is consistent with the above method. The result of the reproducibility analysis can be found in the <xref ref-type="sec" rid="s10">Supplementary Material</xref>.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Result</title>
<sec id="s3-1">
<title>3.1 Overall Workflow</title>
<p>The framework of this study is depicted in <xref ref-type="fig" rid="F1">Figure 1</xref>. In the first stage, two benchmark datasets, including a training dataset and an independent testing dataset, were constructed. In the second stage, five feature extraction methods were utilized to encode each peptide sequence, and then a 550-dimensional feature vector was generated. In the third stage, feature scoring methods and grid search with five-fold cross-validation strategy was used for feature selection. In the fourth stage, multiple machine learning methods were employed to build different models. In the fifth stage, we evaluated the predictive performance of the nine models by using a nested five-fold cross-validation and an independent testing dataset, respectively. Finally, the RF model outperformed other models was selected as the final model, which was implemented into a web server.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The framework of BBPpredict. <bold>(A)</bold>. Dataset Construction. <bold>(B)</bold>. Feature extraction. <bold>(C)</bold>. Feature selection. <bold>(D)</bold>. Model construction. <bold>(E)</bold>. Model evaluation. <bold>(F)</bold>. Web service.</p>
</caption>
<graphic xlink:href="fgene-13-845747-g001.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Performance of Nine Classifiers in Nested Five-Fold Cross-Validation</title>
<p>The performance of the nine predictive models in the nested five-fold cross-validation is shown in <xref ref-type="table" rid="T2">Table 2</xref>, and the ROC curves are illustrated in <xref ref-type="fig" rid="F2">Figure 2A</xref>. For a detailed description of nested five-validation cross-validation, please refer to the <xref ref-type="sec" rid="s10">Supplementary Material</xref>. In <xref ref-type="table" rid="T2">Table 2</xref>, RF model outperformed the other eight machine learning models. All five evaluation metrics reached the highest level. It has an AUC score of 0.9030, ACC value of 81.90%, MCC value of 0.6390, SN value of 79.14% and SP value of 84.66% (see <xref ref-type="table" rid="T2">Table 2</xref>). Moreover, compared with the eight conventional machine learning classifiers, the performance of LSTM is not satisfactory. Except for SP, the values of the other four evaluation metrics of LSTM model were the lowest. The overall performance of traditional machine learning algorithms is generally better than LSTM.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>The prediction performances of different classifiers in nested five-fold cross-validation.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Scoring Method</th>
<th align="center">Classifier</th>
<th align="center">SN(%)</th>
<th align="center">SP(%)</th>
<th align="center">ACC(%)</th>
<th align="center">MCC</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="9" align="left">F-score</td>
<td>
<bold>RF</bold>
</td>
<td align="char" char=".">
<bold>79.14</bold>
</td>
<td align="char" char=".">
<bold>84.66</bold>
</td>
<td align="char" char=".">
<bold>81.90</bold>
</td>
<td align="char" char=".">
<bold>0.6390</bold>
</td>
<td align="char" char=".">
<bold>0.9030</bold>
</td>
</tr>
<tr>
<td align="left">KNN</td>
<td>76.69</td>
<td align="char" char=".">80.98</td>
<td align="char" char=".">78.83</td>
<td align="char" char=".">0.5772</td>
<td align="char" char=".">0.7883</td>
</tr>
<tr>
<td align="left">rbfSVM</td>
<td>78.83</td>
<td align="char" char=".">83.13</td>
<td align="char" char=".">80.98</td>
<td align="char" char=".">0.6202</td>
<td align="char" char=".">0.8872</td>
</tr>
<tr>
<td align="left">linearSVM</td>
<td>75.77</td>
<td align="char" char=".">83.13</td>
<td align="char" char=".">79.45</td>
<td align="char" char=".">0.5906</td>
<td align="char" char=".">0.8690</td>
</tr>
<tr>
<td align="left">DT</td>
<td>71.78</td>
<td align="char" char=".">74.54</td>
<td align="char" char=".">73.16</td>
<td align="char" char=".">0.4634</td>
<td align="char" char=".">0.7357</td>
</tr>
<tr>
<td align="left">LSTM</td>
<td>65.23</td>
<td align="char" char=".">75.38</td>
<td align="char" char=".">70.31</td>
<td align="char" char=".">0.4083</td>
<td align="char" char=".">0.7313</td>
</tr>
<tr>
<td align="left">AdaBoost</td>
<td>77.91</td>
<td align="char" char=".">80.67</td>
<td align="char" char=".">79.29</td>
<td align="char" char=".">0.5861</td>
<td align="char" char=".">0.8615</td>
</tr>
<tr>
<td align="left">GentleBoost</td>
<td>77.30</td>
<td align="char" char=".">80.06</td>
<td align="char" char=".">78.68</td>
<td align="char" char=".">0.5738</td>
<td align="char" char=".">0.8582</td>
</tr>
<tr>
<td align="left">LogitBoost</td>
<td>79.14</td>
<td align="char" char=".">82.21</td>
<td align="char" char=".">80.67</td>
<td align="char" char=".">0.6138</td>
<td align="char" char=".">0.8680</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Performance evaluation of different predictors in five-fold cross-validation and independent testing dataset. <bold>(A)</bold> ROC curves of the five-fold cross-validation. <bold>(B)</bold> ROC curves of the independent testing dataset.</p>
</caption>
<graphic xlink:href="fgene-13-845747-g002.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 Performance of Nine Classifiers on the Independent Testing Dataset</title>
<p>To determine the final model for constructing BBPpredict, performance evaluation on the independent testing dataset is much more convincing than five-fold cross-validation. According to the steps in the method section, nine classification models are established by using the training dataset. The independent testing dataset was then utilized to test the performance of these models. As depicted in <xref ref-type="table" rid="T3">Table 3</xref> and <xref ref-type="fig" rid="F2">Figure 2B</xref>, in term of AUC score, the RF model also performed best, with a score of 0.8332, higher than rbfSVM, linearSVM, KNN, DT, GentleBoost, AdaBoost, LogitBoost and LSTM classifiers by 0.0091, 0.0676, 0.1463, 0.1758, 0.0501, 0.0943, 0.0534 and 0.2291 respectively. In terms of accuracy and MCC, the RF classifier also achieved impressive values, with scores of 77.27% and 0.5455, which are better than other eight classifier algorithm predictors. Furthermore, the LSTM classifier had the weakest generalization ability. In addition, results of the reproducibility analysis for nine classifiers are highly consistent with the above results (see <xref ref-type="sec" rid="s10">Supplementary Table S9)</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The prediction performances of different classifiers in the independent testing dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Scoring Method</th>
<th align="center">Classifier</th>
<th align="center">SN(%)</th>
<th align="center">SP(%)</th>
<th align="center">ACC(%)</th>
<th align="center">MCC</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="9" align="left">F-score</td>
<td>
<bold>RF</bold>
</td>
<td align="char" char=".">
<bold>76.77</bold>
</td>
<td align="char" char=".">
<bold>77.78</bold>
</td>
<td align="char" char=".">
<bold>77.27</bold>
</td>
<td align="char" char=".">
<bold>0.5455</bold>
</td>
<td align="char" char=".">
<bold>0.8332</bold>
</td>
</tr>
<tr>
<td align="left">rbfSVM</td>
<td>78.79</td>
<td align="char" char=".">73.74</td>
<td align="char" char=".">76.26</td>
<td align="char" char=".">0.5259</td>
<td align="char" char=".">0.8241</td>
</tr>
<tr>
<td align="left">KNN</td>
<td>70.71</td>
<td align="char" char=".">66.67</td>
<td align="char" char=".">68.69</td>
<td align="char" char=".">0.3740</td>
<td align="char" char=".">0.6869</td>
</tr>
<tr>
<td align="left">DT</td>
<td>69.70</td>
<td align="char" char=".">61.62</td>
<td align="char" char=".">65.66</td>
<td align="char" char=".">0.3142</td>
<td align="char" char=".">0.6574</td>
</tr>
<tr>
<td align="left">linearSVM</td>
<td>64.65</td>
<td align="char" char=".">74.75</td>
<td align="char" char=".">69.70</td>
<td align="char" char=".">0.3960</td>
<td align="char" char=".">0.7656</td>
</tr>
<tr>
<td align="left">LSTM</td>
<td>58.59</td>
<td align="char" char=".">63.64</td>
<td align="char" char=".">61.11</td>
<td align="char" char=".">0.2225</td>
<td align="char" char=".">0.6041</td>
</tr>
<tr>
<td align="left">AdaBoost</td>
<td>64.65</td>
<td align="char" char=".">68.69</td>
<td align="char" char=".">66.67</td>
<td align="char" char=".">0.3336</td>
<td align="char" char=".">0.7389</td>
</tr>
<tr>
<td align="left">GentleBoost</td>
<td>74.75</td>
<td align="char" char=".">66.67</td>
<td align="char" char=".">70.71</td>
<td align="char" char=".">0.4155</td>
<td align="char" char=".">0.7831</td>
</tr>
<tr>
<td align="left">LogitBoost</td>
<td>67.68</td>
<td align="char" char=".">77.78</td>
<td align="char" char=".">72.73</td>
<td align="char" char=".">0.4569</td>
<td align="char" char=".">0.7798</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-4">
<title>3.4 Performance of the Predictions Under the Combinations of RF With Three Feature Scoring Methods</title>
<p>We also used the RF algorithm with optimal features selected by Pearson and Lasso feature scoring methods to construct prediction model. As shown in <xref ref-type="sec" rid="s10">Supplementary Tables S4,5</xref>, the model under the combination of RF and F-score achieved the second highest AUC value in the nested five-fold cross-validation and the highest AUC value in the independent testing dataset. Therefore, we finally chose the combination of RF and F-score to build the final model based on 184 features and tree depth of 63.</p>
</sec>
<sec id="s3-5">
<title>3.5 Prediction Performance of Existing Predictors</title>
<p>There are two published predictors for identifying BBPs, B3Pred and BBPpred. These predictors and our predictor are based on peptide sequence information. The comparison of datasets of existing predictors and our proposed predictor can be seen in <xref ref-type="table" rid="T4">Table 4</xref> (Detailed comparison can be found in <xref ref-type="sec" rid="s10">Supplementary Table S8</xref>). To be fair, an independent testing dataset, which is completely independent of three predictors&#x2019; training datasets, was used to compare their performance. As shown in <xref ref-type="table" rid="T5">Table 5</xref>, compared with the existing BBPs predictors, our predictor achieved a promising performance (ACC &#x3d; 77.27%, SN &#x3d; 76.77%, SP &#x3d; 77.78% and MCC &#x3d; 0.5455), it outperformed BBPpred and B3Pred, higher than them by 10.6% and 9.59% in accuracy, severally, with MCC increasing 0.2121 and 0.1913, respectively. There were remarkable improvements in sensitivity and specificity (see <xref ref-type="table" rid="T5">Table 5</xref>). The above results demonstrate that BBPpredict is more capable of distinguishing between BBPs and non-BBPs than BBPpred and B3Pred.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Comparison of datasets for three predictors.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">BBPpred</th>
<th align="center">B3Pred</th>
<th align="center">BBPpredict</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">Data source</td>
<td align="left">Positive: Brainpeps, PepBank, articles, SATPdb</td>
<td align="left">Positive: B3Pdb</td>
<td align="left">Positive: Brainpeps, B3Pdb, BBPpred, B3Pred, articles</td>
</tr>
<tr>
<td align="left">Negative: UniProt</td>
<td align="left">Negative: UniProt</td>
<td align="left">Negative: UniProt</td>
</tr>
<tr>
<td align="left">Article search deadline</td>
<td align="left"/>
<td align="left">22 July 2020</td>
<td align="left">Nov. 2021</td>
</tr>
<tr>
<td align="left">Article number</td>
<td align="left">7</td>
<td align="left">271</td>
<td align="left">300</td>
</tr>
<tr>
<td align="left">Positive sample number</td>
<td align="left">119 (training:100, testing: 19)</td>
<td align="left">269 (training:215, testing: 54)</td>
<td align="left">425 (training:326, testing: 99)</td>
</tr>
<tr>
<td align="left">Negative sample number</td>
<td align="left">119 (training:100, testing: 19)</td>
<td align="left">2,690 (training: 2,152, testing:538)</td>
<td align="left">425 (training:326, testing: 99)</td>
</tr>
<tr>
<td align="left">Peptide length</td>
<td align="left">5&#x2013;50</td>
<td align="left">6&#x2013;30</td>
<td align="left">5&#x2013;50</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>The prediction performances of different predictors.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Predictor</th>
<th align="center">SN(%)</th>
<th align="center">SP(%)</th>
<th align="center">ACC(%)</th>
<th align="center">MCC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<bold>BBPpredict</bold>
</td>
<td align="char" char=".">
<bold>76.77</bold>
</td>
<td align="char" char=".">
<bold>77.78</bold>
</td>
<td align="char" char=".">
<bold>77.27</bold>
</td>
<td align="char" char=".">
<bold>0.5455</bold>
</td>
</tr>
<tr>
<td align="left">BBPpred</td>
<td align="char" char=".">67.68</td>
<td align="char" char=".">65.66</td>
<td align="char" char=".">66.67</td>
<td align="char" char=".">0.3334</td>
</tr>
<tr>
<td align="left">B3Pred</td>
<td align="char" char=".">70.71</td>
<td align="char" char=".">64.65</td>
<td align="char" char=".">67.68</td>
<td align="char" char=".">0.3542</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-6">
<title>3.6 Web Server Implementation</title>
<p>To facilitate users to identify BBPs, we established an online web service named BBPpredict that was implemented based on optimized features and the RF model. BBPpredict can be accessed at <ext-link ext-link-type="uri" xlink:href="http://i.uestc.edu.cn/BBPpredict/cgi-bin/BBPpredict.pl">http://i.uestc.edu.cn/BBPpredict/cgi-bin/BBPpredict.pl</ext-link>, conveniently. The web service of BBPpredict was developed by using Perl and Html, <italic>Python</italic> and Matlab. Users can paste peptide sequences or upload a sequence file to predict BBPs, as illustrated in <xref ref-type="fig" rid="F3">Figure 3A</xref>. Then click the &#x201c;Predict&#x201d; button to make predictions, and the predictive results are depicted in <xref ref-type="fig" rid="F3">Figure 3B</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Web interface of BBPpredict. <bold>(A)</bold> The query sequences and threshold of the probability value (tp) are required to be submitted in the input interface. <bold>(B)</bold> The result page returned from BBPpredict.</p>
</caption>
<graphic xlink:href="fgene-13-845747-g003.tif"/>
</fig>
<p>BBPpredict allows users to adjust the threshold of the probability value (tp) to distinguish between predicted positives and negatives, which can range from 0 to 1. As shown in <xref ref-type="table" rid="T6">Table 6</xref>, with the increase of tp, the value of SN decreases, and the SP increases. When tp is 0.5, ACC achieves the highest score of 77.27%, MCC reaches the highest value of 0.5455.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Performance of BBPpredict in the independent testing dataset when tp changes.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">tp</th>
<th align="center">SN (%)</th>
<th align="center">SP (%)</th>
<th align="center">ACC (%)</th>
<th align="center">MCC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">0.1</td>
<td align="char" char=".">100</td>
<td align="char" char=".">11.11</td>
<td align="char" char=".">55.56</td>
<td align="char" char=".">0.2425</td>
</tr>
<tr>
<td align="left">0.2</td>
<td align="char" char=".">98.99</td>
<td align="char" char=".">29.29</td>
<td align="char" char=".">64.14</td>
<td align="char" char=".">0.3944</td>
</tr>
<tr>
<td align="left">0.3</td>
<td align="char" char=".">94.95</td>
<td align="char" char=".">44.44</td>
<td align="char" char=".">69.70</td>
<td align="char" char=".">0.4564</td>
</tr>
<tr>
<td align="left">0.4</td>
<td align="char" char=".">86.87</td>
<td align="char" char=".">64.65</td>
<td align="char" char=".">75.76</td>
<td align="char" char=".">0.5284</td>
</tr>
<tr>
<td align="left">0.5</td>
<td align="char" char=".">76.77</td>
<td align="char" char=".">77.78</td>
<td align="char" char=".">77.27</td>
<td align="char" char=".">0.5455</td>
</tr>
<tr>
<td align="left">0.6</td>
<td align="char" char=".">58.59</td>
<td align="char" char=".">82.83</td>
<td align="char" char=".">70.71</td>
<td align="char" char=".">0.4269</td>
</tr>
<tr>
<td align="left">0.7</td>
<td align="char" char=".">45.45</td>
<td align="char" char=".">90.91</td>
<td align="char" char=".">68.18</td>
<td align="char" char=".">0.4082</td>
</tr>
<tr>
<td align="left">0.8</td>
<td align="char" char=".">36.36</td>
<td align="char" char=".">96.97</td>
<td align="char" char=".">66.67</td>
<td align="char" char=".">0.4191</td>
</tr>
<tr>
<td align="left">0.9</td>
<td align="char" char=".">13.13</td>
<td align="char" char=".">97.98</td>
<td align="char" char=".">55.56</td>
<td align="char" char=".">0.2100</td>
</tr>
<tr>
<td align="left">0.95</td>
<td align="char" char=".">5.05</td>
<td align="char" char=".">97.98</td>
<td align="char" char=".">51.51</td>
<td align="char" char=".">0.0820</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<title>4 Discussion</title>
<p>In the past 30&#xa0;years, many studies have demonstrated that BBPs are promising for the treatment of CNS diseases. BBPs can pass through the BBB and enter brain parenchyma without destroying BBB. Them can be used as transport carriers of DNA, RNA and protein as well as drug-assisted treatment and diagnosis of CNS diseases. However, the discovery of BBPs is still a thorny problem. Only a few hundreds of peptides have been experimentally confirmed as BBPs so far, since BBPs were discovered in 1996 (<xref ref-type="bibr" rid="B2">Banks and Kastin, 1996</xref>). Therefore, to facilitate the treatment of CNS diseases, it is necessary to employ computational methods to rapidly discover and identify more novel BBPs.</p>
<p>At present, two BBPs predictors, BBPpred (<xref ref-type="bibr" rid="B14">Dai et al., 2021</xref>) and B3Pred (<xref ref-type="bibr" rid="B24">Kumar et al., 2021a</xref>), have been proposed. Compared with these two predictors, our developed BBPpredict tool was based on a larger training dataset (as shown in <xref ref-type="table" rid="T4">Table 4</xref>). Besides the difference of the training dataset, a nested cross-validation strategy was utilized in the construction of BBPpredict. For common cross-validation, the model parameters were determined manually, and the accuracy based on the cross-validation would be affected by the artificial selection of model parameters, which usually overestimate the accuracy based on the cross-validation. For nested cross-validation, the model parameters were determined automatically. We speculated that this might be a reason why the previous two predictors had better performance in the cross-validation but had poor performance in our independent testing dataset. BBPpredict showed a large improvement in performance with nearly 6% sensitivity, 12% specificity, 10% accuracy and 0.20 MCC increase, compared with BBPpred and B3Pred. The elevated performance can save cost for researchers to identify BBPs and speed up the discovery of BBPs.</p>
<p>The BBPpredict website allows users to set the tp value. We tested the performance of BBPpredict in the independent testing dataset and provided sensitivity and specificity values under different tp values, which can serve as reference for users and increases the confidence they can have about the positive predictions.</p>
<p>We also reconstructed the BBPs/non-BBPs classification models with different machine learning methods using the new feature vectors that were generated from 16 feature extraction methods, including AAC, DPC, CKSAAGP, PAAC, GAAC, Grouped Di-Peptide Composition (GDPC) (<xref ref-type="bibr" rid="B10">Chen et al., 2018</xref>; <xref ref-type="bibr" rid="B11">Chen et al., 2020</xref>), Dipeptide Deviation from Expected Mean (DDE) (<xref ref-type="bibr" rid="B11">Chen et al., 2020</xref>), Composition (CTDC) (<xref ref-type="bibr" rid="B16">Dubchak et al., 1995</xref>; <xref ref-type="bibr" rid="B17">Dubchak et al., 1999</xref>; <xref ref-type="bibr" rid="B11">Chen et al., 2020</xref>), Transition (CTDT) (<xref ref-type="bibr" rid="B16">Dubchak et al., 1995</xref>; <xref ref-type="bibr" rid="B17">Dubchak et al., 1999</xref>; <xref ref-type="bibr" rid="B11">Chen et al., 2020</xref>), Distribution (CTDD) (<xref ref-type="bibr" rid="B11">Chen et al., 2020</xref>), Amphiphilic Pseudo-Amino Acid Composition (APAAC) (<xref ref-type="bibr" rid="B13">Chou, 2005</xref>; <xref ref-type="bibr" rid="B22">Jiao and Du, 2016</xref>), Quasi-sequence-order (QSOrder) (<xref ref-type="bibr" rid="B11">Chen et al., 2020</xref>), Normalized Moreau-Broto Autocorrelation (NMBroto) (<xref ref-type="bibr" rid="B10">Chen et al., 2018</xref>), Geary correlation (Geary) (<xref ref-type="bibr" rid="B11">Chen et al., 2020</xref>), Moran correlation (Moran) (<xref ref-type="bibr" rid="B18">Feng and Zhang, 2000</xref>; <xref ref-type="bibr" rid="B11">Chen et al., 2020</xref>) and Sequence-Order-Coupling Number (SOCNumber) (<xref ref-type="bibr" rid="B29">Lim et al., 2015</xref>). The detailed description of the last 11 feature encoding approaches can be found in the <xref ref-type="sec" rid="s10">Supplementary Materials</xref>. F-score was used for feature sorting, grid search with five-fold cross-validation was utilized to select the best feature parameters and the best classifier parameters for different classifiers. <xref ref-type="sec" rid="s10">Supplementary Tables S6,7</xref> illustrated the detailed results of five-fold cross-validation and independent testing dataset of reconstructed classification models, respectively. However, the addition of feature encoding methods did not improve the classification performance of the model. We speculate that it is caused by the high correlation between the extracted features based on different feature extracting methods, which might induce highly correlated features in the final feature subset. As the feature number is limited, the highly correlated features might reduce useful information for model construction. Another possible reason might be the limited sample size, which might cause high false positive rate during the process of feature selection. The increase of feature size would lead to the increase of false positive features, which would affect the robustness of the predictive model.</p>
<p>BBPs pass through BBB via six penetration mechanisms, including diffusion transport, carrier-mediated transcytosis, efflux transporter, receptor-mediated transcytosis, adsorptive-mediated transcytosis and cell-mediated transcytosis (<xref ref-type="bibr" rid="B44">Zhou et al., 2021</xref>). The abilities of BBPs to penetrate BBB vary depending on their penetration mechanisms (<xref ref-type="bibr" rid="B34">S&#xe1;nchez-Navarro et al., 2017</xref>). Therefore, we speculate the differences in their penetration mechanisms may affect the reliability of screening in the procession of model construction. However, BBPs of distinct penetration mechanisms were not further divided when constructing the positive sample of BBPpred, B3Pred and BBPpredict, because the number of BBPs for a specific transport mechanism is insufficient to construct a BBP predictor.</p>
<p>In the present work, we utilized RF algorithm to construct BBP predictor. The RF is an ensemble algorithm which is composed of several weak classifiers (decision trees). Our constructed model contains 63 decision trees. We speculate that these different decision trees might cover different penetration mechanisms and it might be the reason why the RF algorithm is superior to other machine learning algorithms. In the future, if the number of BBPs with a certain transport mechanism increase, it is possible and preferable to construct new BBP predictors using BBPs with the same penetrating mechanism.</p>
</sec>
<sec id="s5">
<title>5 Conclusion</title>
<p>In this study, we proposed an RF-based predictor for identifying BBPs, called BBPpredict, which is available for free at <ext-link ext-link-type="uri" xlink:href="http://i.uestc.edu.cn/BBPpredict/cgi-bin/BBPpredict.pl">http://i.uestc.edu.cn/BBPpredict/cgi-bin/BBPpredict.pl</ext-link>. To find the optimal classifier, eight traditional machine learning algorithms and one deep learning algorithm were used for developing models. The RF algorithm was selected to construct BBPpredict after comparing the results of nine classifiers in the five-fold cross-validation and independent test. The RF-based model reached an AUC of 0.9030 with an accuracy of 81.90% and an AUC of 0.8332 with an accuracy of 77.27% in the nested five-fold cross-validation and independent testing dataset, respectively. We also compared BBPpredict with two existing BBPs predictors, BBPpred and B3Pred. The results showed that BBPpredict was remarkably higher in accuracy, MCC, sensitivity and specificity than these two predictors. BBPpredict is a promising classification model, and we expect it to play a positive role in the discovery of BBPs to facilitate the development of drugs for CNS diseases.</p>
</sec>
</body>
<back>
<sec id="s6">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s10">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>XC, QZ, BL, CL, SY, JL, BH, HC, and JH developed the web interface of the predictor. XC conceived and designed the experiments, performed the experiments, analyzed the data, prepared figures and/or tables, authored or reviewed drafts of the paper, and approved the final draft. BH, HC, and JH conceived and designed the experiments, authored or reviewed drafts of the paper. All authors approved the final draft.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This work was supported by the National Natural Science Foundation of China (grant numbers: 61901130, 61901129, and 62071099), Science and Technology Department of Guizhou Province (Grant Numbers: (2020)1Y407 and ZK[2022]-General-038) and Guizhou University (Grant Numbers: (2018)54, (2018)55 and (2020)5).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>The authors are grateful to the reviewers for their valuable suggestions and comments, which will lead to the improvement of this paper.</p>
</ack>
<sec id="s11">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.845747/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.845747/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Banks</surname>
<given-names>W. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>From Blood-Brain Barrier to Blood-Brain Interface: New Opportunities for CNS Drug Delivery</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>15</volume> (<issue>4</issue>), <fpage>275</fpage>&#x2013;<lpage>292</lpage>. <pub-id pub-id-type="doi">10.1038/nrd.2015.21</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Banks</surname>
<given-names>W. A.</given-names>
</name>
<name>
<surname>Kastin</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Passage of Peptides across the Blood-Brain Barrier: Pathophysiological Perspectives</article-title>. <source>Life Sci.</source> <volume>59</volume> (<issue>23</issue>), <fpage>1923</fpage>&#x2013;<lpage>1943</lpage>. <pub-id pub-id-type="doi">10.1016/s0024-3205(96)00380-3</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhadra</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fong</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Siu</surname>
<given-names>S. W. I.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>AmPEP: Sequence-Based Prediction of Antimicrobial Peptides Using Distribution Patterns of Amino Acid Properties and Random forest</article-title>. <source>Sci. Rep.</source> <volume>8</volume> (<issue>1</issue>), <fpage>1697</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-19752-w</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhasin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Raghava</surname>
<given-names>G. P. S.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Classification of Nuclear Receptors Based on Amino Acid Composition and Dipeptide Composition</article-title>. <source>J. Biol. Chem.</source> <volume>279</volume> (<issue>22</issue>), <fpage>23262</fpage>&#x2013;<lpage>23266</lpage>. <pub-id pub-id-type="doi">10.1074/jbc.M401932200</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname>
<given-names>C.-C.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>C.-J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>LIBSVM: A Library for Support Vector Machines</article-title>. <source>ACM Trans. Intell. Syst. Technol.</source> <volume>2</volume> (<issue>3</issue>), <fpage>1</fpage>&#x2013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1145/1961189.1961199</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kurgan</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Prediction of Integral Membrane Protein Type by Collocated Hydrophobic Amino Acid Pairs</article-title>. <source>J. Comput. Chem.</source> <volume>30</volume> (<issue>1</issue>), <fpage>163</fpage>&#x2013;<lpage>172</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.21053</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kurgan</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Ruan</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2007b</year>). <article-title>Prediction of Flexible/rigid Regions from Protein Sequences Using K-Spaced Amino Acid Pairs</article-title>. <source>BMC Struct. Biol.</source> <volume>7</volume>, <fpage>25</fpage>. <pub-id pub-id-type="doi">10.1186/1472-6807-7-25</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kurgan</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Ruan</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Prediction of Protein Structural Class Using Novel Evolutionary Collocation-Based Sequence Representation</article-title>. <source>J. Comput. Chem.</source> <volume>29</volume> (<issue>10</issue>), <fpage>1596</fpage>&#x2013;<lpage>1604</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.20918</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kurgan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rahbari</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2007a</year>). <article-title>Prediction of Protein Crystallization Using Collocation of Amino Acid Pairs</article-title>. <source>Biochem. Biophysical Res. Commun.</source> <volume>355</volume> (<issue>3</issue>), <fpage>764</fpage>&#x2013;<lpage>769</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbrc.2007.02.040</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Leier</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Marquez-Lago</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>iFeature: a Python Package and Web Server for Features Extraction and Selection from Protein and Peptide Sequences</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>14</issue>), <fpage>2499</fpage>&#x2013;<lpage>2502</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty140</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Marquez-Lago</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>Leier</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Revote</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>iLearn: an Integrated Platform and Meta-Learner for Feature Engineering, Machine-Learning Analysis and Modeling of DNA, RNA and Protein Sequence Data</article-title>. <source>Brief Bioinform</source> <volume>21</volume> (<issue>3</issue>), <fpage>1047</fpage>&#x2013;<lpage>1057</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz041</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Prediction of Protein Cellular Attributes Using Pseudo-amino Acid Composition</article-title>. <source>Proteins</source> <volume>43</volume> (<issue>3</issue>), <fpage>246</fpage>&#x2013;<lpage>255</lpage>. <pub-id pub-id-type="doi">10.1002/prot.1035</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Using Amphiphilic Pseudo Amino Acid Composition to Predict Enzyme Subfamily Classes</article-title>. <source>Bioinformatics</source> <volume>21</volume> (<issue>1</issue>), <fpage>10</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bth466</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wynendaele</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Bin</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>BBPpred: Sequence-Based Prediction of Blood-Brain Barrier Peptides with Feature Representation Learning and Logistic Regression</article-title>. <source>J. Chem. Inf. Model.</source> <volume>61</volume> (<issue>1</issue>), <fpage>525</fpage>&#x2013;<lpage>534</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.0c01115</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Drappatz</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Brenner</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>E. T.</given-names>
</name>
<name>
<surname>Eichler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schiff</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Groves</surname>
<given-names>M. D.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Phase I Study of GRN1005 in Recurrent Malignant Glioma</article-title>. <source>Clin. Cancer Res.</source> <volume>19</volume> (<issue>6</issue>), <fpage>1567</fpage>&#x2013;<lpage>1576</lpage>. <pub-id pub-id-type="doi">10.1158/1078-0432.Ccr-12-2481</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dubchak</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Muchnik</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Holbrook</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S. H.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Prediction of Protein Folding Class Using Global Description of Amino Acid Sequence</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>92</volume> (<issue>19</issue>), <fpage>8700</fpage>&#x2013;<lpage>8704</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.92.19.8700</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dubchak</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Muchnik</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Mayor</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dralyuk</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.-H.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Recognition of a Protein Fold in the Context of the SCOP Classification</article-title>. <source>Proteins</source> <volume>35</volume> (<issue>4</issue>), <fpage>401</fpage>&#x2013;<lpage>407</lpage>. <pub-id pub-id-type="doi">10.1002/(sici)1097-0134(19990601)35:4&#x3c;401::aid-prot3&#x3e;3.0.co;2-k</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>Z.-P.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.-T.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Prediction of Membrane Protein Types Based on the Hydrophobic index of Amino Acids</article-title>. <source>J. Protein Chem.</source> <volume>19</volume> (<issue>4</issue>), <fpage>269</fpage>&#x2013;<lpage>275</lpage>. <pub-id pub-id-type="doi">10.1023/a:1007091128394</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>PhD7Faster 2.0: Predicting Clones Propagating Faster from the Ph.D.-7 Phage Display Library by Coupling PseAAC and Tripeptide Composition</article-title>. <source>PeerJ</source> <volume>7</volume>, <fpage>e7131</fpage>. <pub-id pub-id-type="doi">10.7717/peerj.7131</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ru</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>SABinder: A Web Service for Predicting Streptavidin-Binding Peptides</article-title>. <source>Biomed. Res. Int.</source> <volume>2016</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1155/2016/9175143</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hochreiter</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Schmidhuber</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Long Short-Term Memory</article-title>. <source>Neural Comput.</source> <volume>9</volume> (<issue>8</issue>), <fpage>1735</fpage>&#x2013;<lpage>1780</lpage>. <pub-id pub-id-type="doi">10.1162/neco.1997.9.8.1735</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiao</surname>
<given-names>Y.-S.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>P.-F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Predicting Golgi-Resident Protein Types Using Pseudo Amino Acid Compositions: Approaches with Positional Specific Physicochemical Properties</article-title>. <source>J. Theor. Biol.</source> <volume>391</volume>, <fpage>35</fpage>&#x2013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2015.11.009</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bhalla</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Usmani</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Varshney</surname>
<given-names>G. C.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Prediction of Cell-Penetrating Potential of Modified Peptides Containing Natural and Chemically Modified Residues</article-title>. <source>Front. Microbiol.</source> <volume>9</volume>, <fpage>725</fpage>. <pub-id pub-id-type="doi">10.3389/fmicb.2018.00725</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Patiyal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dhall</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Raghava</surname>
<given-names>G. P. S.</given-names>
</name>
</person-group> (<year>2021a</year>). <article-title>B3Pred: A Random-Forest-Based Method for Predicting and Designing Blood-Brain Barrier Penetrating Peptides</article-title>. <source>Pharmaceutics</source> <volume>13</volume> (<issue>8</issue>), <fpage>1237</fpage>. <pub-id pub-id-type="doi">10.3390/pharmaceutics13081237</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Patiyal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sahai</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kaur</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lathwal</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2021b</year>). <article-title>B3Pdb: an Archive of Blood-Brain Barrier-Penetrating Peptides</article-title>. <source>Brain Struct. Funct.</source> <volume>226</volume> (<issue>8</issue>), <fpage>2489</fpage>&#x2013;<lpage>2495</lpage>. <pub-id pub-id-type="doi">10.1007/s00429-021-02341-5</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kurzrock</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gabrail</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Chandhasin</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Moulder</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Brenner</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Safety, Pharmacokinetics, and Activity of GRN1005, a Novel Conjugate of Angiopep-2, a Peptide Facilitating Brain Penetration, and Paclitaxel, in Patients with Advanced Solid Tumors</article-title>. <source>Mol. Cancer Ther.</source> <volume>11</volume> (<issue>2</issue>), <fpage>308</fpage>&#x2013;<lpage>316</lpage>. <pub-id pub-id-type="doi">10.1158/1535-7163.Mct-11-0566</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>T.-Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Z.-Q.</given-names>
</name>
<name>
<surname>Hsieh</surname>
<given-names>S.-J.</given-names>
</name>
<name>
<surname>Breta&#xf1;a</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>C.-T.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Exploiting Maximal Dependence Decomposition to Identify Conserved Motifs from a Group of Aligned Signal Sequences</article-title>. <source>Bioinformatics</source> <volume>27</volume> (<issue>13</issue>), <fpage>1780</fpage>&#x2013;<lpage>1787</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr291</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>F.-M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.-Q.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Identifying Anticancer Peptides by Using Improved Hybrid Compositions</article-title>. <source>Sci. Rep.</source> <volume>6</volume>, <fpage>33910</fpage>. <pub-id pub-id-type="doi">10.1038/srep33910</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>W.-J.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Koo</surname>
<given-names>J.-H.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.-A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>dNP2 Is a Blood-Brain Barrier-Permeable Peptide Enabling ctCTLA-4 Protein Delivery to Ameliorate Experimental Autoimmune Encephalomyelitis</article-title>. <source>Nat. Commun.</source> <volume>6</volume>, <fpage>8244</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms9244</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muttenthaler</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>King</surname>
<given-names>G. F.</given-names>
</name>
<name>
<surname>Adams</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Alewood</surname>
<given-names>P. F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Trends in Peptide Drug Discovery</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>20</volume> (<issue>4</issue>), <fpage>309</fpage>&#x2013;<lpage>325</lpage>. <pub-id pub-id-type="doi">10.1038/s41573-020-00135-8</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nance</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pun</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Saigal</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sellers</surname>
<given-names>D. L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Drug Delivery to the central Nervous System</article-title>. <source>Nat. Rev. Mater</source> <volume>7</volume> (<issue>4</issue>), <fpage>314</fpage>&#x2013;<lpage>331</lpage>. <pub-id pub-id-type="doi">10.1038/s41578-021-00394-w</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nonaka</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Suzuki-Anekoji</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nakayama</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mabashi-Asazuma</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jarvis</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Yeh</surname>
<given-names>J.-C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Overcoming the Blood-Brain Barrier by Annexin A1-Binding Peptide to Target Brain Tumours</article-title>. <source>Br. J. Cancer</source> <volume>123</volume> (<issue>11</issue>), <fpage>1633</fpage>&#x2013;<lpage>1643</lpage>. <pub-id pub-id-type="doi">10.1038/s41416-020-01066-2</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oller-Salvia</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>S&#xe1;nchez-Navarro</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Giralt</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Teixid&#xf3;</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Blood-brain Barrier Shuttle Peptides: an Emerging Paradigm for Brain Delivery</article-title>. <source>Chem. Soc. Rev.</source> <volume>45</volume> (<issue>17</issue>), <fpage>4690</fpage>&#x2013;<lpage>4707</lpage>. <pub-id pub-id-type="doi">10.1039/c6cs00076b</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>S&#xe1;nchez-Navarro</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Giralt</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Teixid&#xf3;</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Blood-brain Barrier Peptide Shuttles</article-title>. <source>Curr. Opin. Chem. Biol.</source> <volume>38</volume>, <fpage>134</fpage>&#x2013;<lpage>140</lpage>. <pub-id pub-id-type="doi">10.1016/j.cbpa.2017.04.019</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saravanan</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Gautham</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Harnessing Computational Biology for Exact Linear B-Cell Epitope Prediction: A Novel Amino Acid Composition-Based Feature Descriptor</article-title>. <source>OMICS: A J. Integr. Biol.</source> <volume>19</volume> (<issue>10</issue>), <fpage>648</fpage>&#x2013;<lpage>658</lpage>. <pub-id pub-id-type="doi">10.1089/omi.2015.0095</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Terstappen</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Meyer</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Bell</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Strategies for Delivering Therapeutics across the Blood-Brain Barrier</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>20</volume> (<issue>5</issue>), <fpage>362</fpage>&#x2013;<lpage>383</lpage>. <pub-id pub-id-type="doi">10.1038/s41573-021-00139-y</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Dorpe</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bronselaer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nielandt</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stalmans</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wynendaele</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Audenaert</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Brainpeps: the Blood-Brain Barrier Peptide Database</article-title>. <source>Brain Struct. Funct.</source> <volume>217</volume> (<issue>3</issue>), <fpage>687</fpage>&#x2013;<lpage>718</lpage>. <pub-id pub-id-type="doi">10.1007/s00429-011-0375-0</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Walzer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vizca&#xed;no</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Review of Issues and Solutions to Data Analysis Reproducibility and Data Quality in Clinical Proteomics</article-title>. <source>Methods Mol. Biol.</source> <volume>2051</volume>, <fpage>345</fpage>&#x2013;<lpage>371</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-4939-9744-2_15</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Marquez-Lago</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>Leier</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Bastion3: a Two-Layer Ensemble Predictor of Type III Secreted Effectors</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>12</issue>), <fpage>2017</fpage>&#x2013;<lpage>2028</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty914</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2017a</year>). <article-title>SkipCPP-Pred: an Improved and Promising Sequence-Based Predictor for Predicting Cell-Penetrating Peptides</article-title>. <source>BMC Genomics</source> <volume>18</volume> (<issue>Suppl. 7</issue>), <fpage>742</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-017-4128-1</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Z. S.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2017b</year>). <article-title>CPPred-RF: A Sequence-Based Predictor for Identifying Cell-Penetrating Peptides and Their Uptake Efficiency</article-title>. <source>J. Proteome Res.</source> <volume>16</volume> (<issue>5</issue>), <fpage>2044</fpage>&#x2013;<lpage>2053</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jproteome.7b00019</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Retro-enantio Isomer of Angiopep-2 Assists Nanoprobes across the Blood-Brain Barrier for Targeted Magnetic Resonance/fluorescence Imaging of Glioblastoma</article-title>. <source>Sig Transduct Target. Ther.</source> <volume>6</volume> (<issue>1</issue>), <fpage>309</fpage>. <pub-id pub-id-type="doi">10.1038/s41392-021-00724-y</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Y.-W.</given-names>
</name>
<name>
<surname>Lai</surname>
<given-names>H.-Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Prediction of Phosphothreonine Sites in Human Proteins by Fusing Different Features</article-title>. <source>Sci. Rep.</source> <volume>6</volume>, <fpage>34817</fpage>. <pub-id pub-id-type="doi">10.1038/srep34817</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>Q. R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Brain Penetrating Peptides and Peptide-Drug Conjugates to Overcome the Blood-Brain Barrier and Target CNS Diseases</article-title>. <source>WIREs Nanomed Nanobiotechnol</source> <volume>13</volume> (<issue>4</issue>), <fpage>e1695</fpage>. <pub-id pub-id-type="doi">10.1002/wnan.1695</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>