<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mol. Biosci.</journal-id>
<journal-title>Frontiers in Molecular Biosciences</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mol. Biosci.</abbrev-journal-title>
<issn pub-type="epub">2296-889X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1238509</article-id>
<article-id pub-id-type="doi">10.3389/fmolb.2023.1238509</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Molecular Biosciences</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Hierarchical machine learning model predicts antimicrobial peptide activity against <italic>Staphylococcus aureus</italic>
</article-title>
<alt-title alt-title-type="left-running-head">Khabaz et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fmolb.2023.1238509">10.3389/fmolb.2023.1238509</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Khabaz</surname>
<given-names>Hosein</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2325437/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Rahimi-Nasrabadi</surname>
<given-names>Mehdi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Keihan</surname>
<given-names>Amir Homayoun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1766207/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Molecular Biology Research Center</institution>, <institution>Systems Biology and Poisonings Institute</institution>, <institution>Baqiyatallah University of Medical Sciences</institution>, <addr-line>Tehran</addr-line>, <country>Iran</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Faculty of Pharmacy</institution>, <institution>Baqiyatallah University of Medical Sciences</institution>, <addr-line>Tehran</addr-line>, <country>Iran</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1249428/overview">Cigdem Sevim Bayrak</ext-link>, Icahn School of Medicine at Mount Sinai, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/184313/overview">Micha&#x142; Burdukiewicz</ext-link>, University of Wroc&#x142;aw, Poland</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2355865/overview">Fernando Lobo Palacios</ext-link>, University of La Laguna, Spain</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/449350/overview">Jian Huang</ext-link>, University of Electronic Science and Technology of China, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Amir Homayoun Keihan, <email>ahkeihan@bmsu.ac.ir</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>18</day>
<month>09</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>10</volume>
<elocation-id>1238509</elocation-id>
<history>
<date date-type="received">
<day>11</day>
<month>06</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>08</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Khabaz, Rahimi-Nasrabadi and Keihan.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Khabaz, Rahimi-Nasrabadi and Keihan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Introduction:</bold> <italic>Staphylococcus aureus</italic> is a dangerous pathogen which causes a vast selection of infections. Antimicrobial peptides have been demonstrated as a new hope for developing antibiotic agents against multi-drug-resistant bacteria such as <italic>S. aureus</italic>. Yet, most studies on developing classification tools for antimicrobial peptide activities do not focus on any specific species, and therefore, their applications are limited.</p>
<p>
<bold>Methods:</bold> Here, by using an up-to-date dataset, we have developed a hierarchical machine learning model for classifying peptides with antimicrobial activity against <italic>S. aureus</italic>. The first-level model classifies peptides into AMPs and non-AMPs. The second-level model classifies AMPs into those active against <italic>S. aureus</italic> and those not active against this species.</p>
<p>
<bold>Results:</bold> Results from both classifiers demonstrate the effectiveness of the hierarchical approach. A comprehensive set of physicochemical and linguistic-based features has been used, and after feature selection steps, only some physicochemical properties were selected. The final model showed the F1-score of 0.80, recall of 0.86, balanced accuracy of 0.80, and specificity of 0.73 on the test set.</p>
<p>
<bold>Discussion:</bold> The susceptibility to a single AMP is highly varied among different target species. Therefore, it cannot be concluded that AMP candidates suggested by AMP/non-AMP classifiers are able to show suitable activity against a specific species. Here, we addressed this issue by creating a hierarchical machine learning model which can be used in practical applications for extracting potential antimicrobial peptides against <italic>S. aureus</italic> from peptide libraries.</p>
</abstract>
<kwd-group>
<kwd>
<italic>Staphylococcus aureus</italic>
</kwd>
<kwd>antimicrobial peptides</kwd>
<kwd>machine learning</kwd>
<kwd>antimicrobial activity</kwd>
<kwd>classification model</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Biological Modeling and Simulation</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>
<italic>Staphylococcus aureus</italic> is a prominent human pathogen that causes a wide range of infections, including pleuropulmonary, osteoarticular, skin, and soft tissue infections. In the United States, nearly 50 percent of deaths caused by antibiotic-resistant bacterial pathogens are attributed to <italic>methicillin-resistant S. aureus</italic> (MRSA) infections (<xref ref-type="bibr" rid="B15">Stryjewski and Chambers, 2008</xref>; <xref ref-type="bibr" rid="B12">Mohamed et al., 2016</xref>). It has been reported that <italic>S. aureus</italic> was the leading bacterial cause of death in 135 countries and had the highest mortality rate in individuals aged above 15&#xa0;years worldwide (<xref ref-type="bibr" rid="B9">Ikuta et al., 2022</xref>). The evasion of staphylococcal biofilms and toxins can result in prolonged inflammation, chronic infections, and delayed wound healing (<xref ref-type="bibr" rid="B21">Wolcott et al., 2010</xref>).</p>
<p>Antimicrobial peptides (AMPs) are a group of naturally occurring or synthetic short peptides with the ability to kill bacterial cells. AMPs typically interact with bacterial membranes, making it difficult for bacteria to develop resistance against them (<xref ref-type="bibr" rid="B23">Zasloff, 2002</xref>; <xref ref-type="bibr" rid="B7">Hancock and Sahl, 2006</xref>). This has generated increased interest in AMPs as potential substitutes for conventional antibiotics as antibiotic resistance has become a global crisis (<xref ref-type="bibr" rid="B7">Hancock and Sahl, 2006</xref>; <xref ref-type="bibr" rid="B3">Chen and Lu, 2020</xref>).</p>
<p>Numerous experimental reports have identified AMPs with antimicrobial activity against <italic>S. aureus</italic>, which are accessible in online databases. However, experimental studies on peptides are often costly and time-consuming (<xref ref-type="bibr" rid="B11">Lee et al., 2017</xref>). In contrast, computational approaches, particularly machine learning techniques, have enabled the development of high-throughput models for predicting various aspects of AMP functionality, including antimicrobial activity (<xref ref-type="bibr" rid="B19">Vishnepolsky and Pirtskhalava, 2014</xref>; <xref ref-type="bibr" rid="B11">Lee et al., 2017</xref>) and toxicity against human cells (<xref ref-type="bibr" rid="B2">Chaudhary et al., 2016</xref>; <xref ref-type="bibr" rid="B10">Kleandrova et al., 2016</xref>).</p>
<p>Most studies in the field of AMP classification have primarily focused on distinguishing AMPs from non-AMPs (<xref ref-type="bibr" rid="B19">Vishnepolsky and Pirtskhalava, 2014</xref>; <xref ref-type="bibr" rid="B10">Kleandrova et al., 2016</xref>). However, due to the significant diversity among target bacteria, the functionality of AMPs could vary greatly among different bacterial families. Therefore, simply predicting a peptide as an AMP does not necessarily imply that it will exhibit activity against a specific bacterial family. This limitation hinders the applicability of general studies in real-world challenges. <xref ref-type="bibr" rid="B17">Vishnepolsky et al. (2018)</xref> developed predictive models to classify AMPs active against Gram-negative bacteria, specifically <italic>Escherichia coli</italic>. Speck-Planche et al. introduced a multi-target model to identify AMPs active against Gram-positive pathogens, including <italic>S. aureus</italic> (<xref ref-type="bibr" rid="B14">Speck-Planche et al., 2016</xref>). However, both studies predominantly utilized datasets composed of strong and weak AMPs. Since AMPs represent a small fraction of the peptide space, models trained solely on this limited portion have limited applicability to peptides that deviate significantly from AMP characteristics.</p>
<p>A hierarchical machine learning model refers to a classification approach that involves a multi-level process of classification, where the output of one classifier serves as the input for another (<xref ref-type="bibr" rid="B20">Wehrmann et al., 2018</xref>). Hierarchical classifiers have been previously used for several function prediction models such as gene function prediction (<xref ref-type="bibr" rid="B16">Valentini, 2010</xref>) and protein function prediction (<xref ref-type="bibr" rid="B4">Eisner et al., 2005</xref>). Here, we use machine learning approaches to construct a hierarchical model that discriminates AMPs with specific antimicrobial activity against <italic>S. aureus</italic> by incorporating physicochemical and linguistic-based properties of peptides.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Preparing data</title>
<p>Two datasets were needed for this study. Dataset-1 includes AMP (positive) and non-AMP (negative) sets. The positive set was created using AMP records from the Database of Antimicrobial Activity and Structure of Peptides (DBAASP) (<xref ref-type="bibr" rid="B6">Gogoladze et al., 2014</xref>). Records with D-amino acids, unnatural residues, C-terminal modifications (except the amid group), and N-terminal modifications (except for acetyl) were removed from the dataset. Moreover, peptide sequences shorter than six residues and longer than 50 residues were also removed since there was no sufficient data in those ranges. All concentrations reported with the &#x3bc;g/mL unit were converted to &#x3bc;M using peptide molecular weight. Only records with a reported minimum inhibitory concentration (MIC) &#x3d;&#x3c; 15&#xa0;&#x3bc;M were used in the final dataset. As for the negative set, the dataset created by <xref ref-type="bibr" rid="B5">Gabere and Noble (2017)</xref> was used with the same restrictions. Highly similar sequences (maximum 90%) were removed using CD-HIT software (<xref ref-type="bibr" rid="B8">Huang et al., 2010</xref>). The final dataset included 2,144 AMPs and 2,144 non-AMPs. The dataset was split into three sets of train (64%), validation (16%), and test (20%) with no overlapping records and enabled stratify argument on the class label.</p>
<p>Dataset 2 included AMP records with reported activity against <italic>S. aureus</italic> from DBAASP. Peptides with an MIC of 10&#xa0;&#xb5;M or lower were labeled as positive, and peptides with an MIC of 15&#xa0;&#xb5;M or higher were labeled as negative. Most AMPs have more than one reported activity in the database. After labeling every record, peptides for which all their records had the same label were kept in the dataset. In other words, if a peptide had mixed positive and negative labels in its activity records, it would be excluded from the dataset. The final dataset 2 included 2,488 positive and 1,595 negative AMPs. Train, validation, and test sets were constructed as mentioned previously.</p>
</sec>
<sec id="s2-2">
<title>2.2 Feature extraction</title>
<p>For dataset 1, a total of seven physicochemical properties, namely, hydrophobicity, net charge, molecular weight, charge density, isoelectric point, hydrophobic moment, and aggregation propensity <italic>in vivo</italic>, were extracted. For dataset 2, a more complex set of features was extracted. A total of 1,527 physicochemical and linguistic-based properties, including autocorrelation, physiochemical composition, transition, and distribution, were extracted from AMP sequences using Propy Python library (<xref ref-type="bibr" rid="B1">Cao et al., 2013</xref>) (<xref ref-type="sec" rid="s10">Supplementary Table S1</xref>). All peptide records from DBAASP have four properties of net charge, normalized hydrophobic moment, normalized hydrophobicity, and isoelectric point. Charge densities were calculated similar to the previous work (<xref ref-type="bibr" rid="B19">Vishnepolsky and Pirtskhalava, 2014</xref>) using net charge and molecular weight.</p>
</sec>
<sec id="s2-3">
<title>2.3 Feature selection strategy</title>
<p>Feature selection was carried out in dataset 2 due to its high number of features. First, using Mathematica software (<xref ref-type="bibr" rid="B22">Wolfram Research, 2020</xref>), the Pearson correlation coefficient between all pairs of features was calculated, and then, features with a correlation over 95% were grouped together. From each set of the correlated group, only one feature was kept. To select relevant features, the SelectFromModel meta-transformer of scikit-learn library (<xref ref-type="bibr" rid="B13">Pedregosa et al., 2011</xref>) was used. The training set was split into five parts with no overlapping records. Then, random forest classifiers were trained on each part independently, and important features were extracted from each classifier. Features shared among all five parts were considered important and kept for further usage.</p>
</sec>
<sec id="s2-4">
<title>2.4 Training classification models</title>
<p>Classification models were trained in datasets 1 and 2 independently. The test step was also carried out separately. Several learning algorithms, including random forest, support vector classification (SVC), linear SVC, K-nearest neighbors, and na&#xef;ve Bayes, were trained using 10-fold cross validation and grid search to optimize hyper-parameters for each algorithm. All algorithms were optimized to obtain the highest F1-scores and the best performance on the validation set. Multiple combinations of hybrid-voting classifiers were also constructed using the best obtained models in dataset 2. The performance of the final models was evaluated on the test set. The comparison of performances was carried out using performance measures including precision, recall, F1-score, accuracy, AUC, and hamming distance.</p>
<p>To evaluate the performance of the final package, a new dataset including AMPs and non-AMPs was used as an independent test. New AMP activity data against <italic>S. aureus</italic> were obtained from DBAASP and processed as mentioned in Methods (such as natural residues, specific sequence length, and no terminal modifications). All AMPs were introduced in 2022&#x2013;2023 and were new for the model. Finally, a total of 118 new peptides including 59 non-AMPs and 59 AMPs with reported activity against <italic>S. aureus</italic> were tested. For final performance measurements, non-AMPs and AMPs not active against <italic>S. aureus</italic> were considered &#x201c;negative,&#x201d; and AMPs active against <italic>S. aureus</italic> were considered &#x201c;positive.&#x201d;</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Susceptibility comparison</title>
<p>The susceptibility of <italic>S. aureus</italic>, <italic>P. aeruginosa</italic>, and <italic>E. coli</italic> to 1,398 AMPs was investigated using the MIC values reported in the DBAASP dataset. Results are shown in <xref ref-type="table" rid="T1">Table 1</xref>. It can be seen that more than 28% of AMPs have more than 20&#xa0;&#xb5;M difference in the MIC against <italic>S. aureus</italic> and <italic>P. aeruginosa</italic>. Even <italic>E. coli</italic> and <italic>P. aeruginosa</italic>, which are both Gram-negative, show very different results. If we use these MIC values to label these AMPs with high antimicrobial activity and low antimicrobial activity labels, the results are going to be highly varied for different species. More than 32% of AMPs obtained opposite labels for <italic>P. aeruginosa</italic> and <italic>S. aureus</italic>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Comparison of susceptibility of <italic>S. aureus</italic>, <italic>P. aeruginosa</italic>, and <italic>E. coli</italic> to similar AMPs.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Comparison</th>
<th align="center">Fraction of AMPs with more than 10&#xa0;&#xb5;M difference in MIC</th>
<th align="center">Fraction of AMPs with more than 20&#xa0;&#xb5;M difference in MIC</th>
<th align="center">Fraction of AMPs with more than 50&#xa0;&#xb5;M difference in MIC</th>
<th align="center">Hamming distance</th>
<th align="center">Percentage of AMPs with opposite activity label (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">
<italic>S. aureus</italic> vs</td>
<td rowspan="2" align="center">0.4134</td>
<td rowspan="2" align="center">0.2847</td>
<td rowspan="2" align="center">0.118</td>
<td rowspan="2" align="center">452</td>
<td rowspan="2" align="center">32.33</td>
</tr>
<tr>
<td align="center">
<italic>P. aeruginosa</italic>
</td>
</tr>
<tr>
<td align="center">
<italic>S. aureus</italic> vs</td>
<td rowspan="2" align="center">0.3376</td>
<td rowspan="2" align="center">0.2310</td>
<td rowspan="2" align="center">0.0937</td>
<td rowspan="2" align="center">384</td>
<td rowspan="2" align="center">27.47</td>
</tr>
<tr>
<td align="center">
<italic>E. coli</italic>
</td>
</tr>
<tr>
<td align="center">
<italic>P. aeruginosa</italic> vs</td>
<td rowspan="2" align="center">0.3541</td>
<td rowspan="2" align="center">0.2389</td>
<td rowspan="2" align="center">0.0973</td>
<td rowspan="2" align="center">380</td>
<td rowspan="2" align="center">27.18</td>
</tr>
<tr>
<td align="center">
<italic>E. coli</italic>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-2">
<title>3.2 Performance of the AMP/non-AMP classification</title>
<p>The first classification model in the pipeline was trained using random forest, SVC, linear SVC, K-nearest neighbors, and na&#xef;ve Bayes algorithm. Performance on the train set can be evaluated by ROC curves (<xref ref-type="fig" rid="F1">Figure 1</xref>). The dotted line shows the performance of a completely random classifier. The larger area under curve in the ROC curve corresponds to a better performance. Models with random forest, SVC, and KNN show very good performances compared to na&#xef;ve Bayes. The performance of models on test sets was evaluated (<xref ref-type="table" rid="T2">Table 2</xref>). The results show that SVC, random forest, and KNN with accuracy and the F1-score above 0.9 show great performances for classifying AMPs and non-AMPs.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>ROC curves obtained for each algorithm classifying AMPs from non-AMPs.</p>
</caption>
<graphic xlink:href="fmolb-10-1238509-g001.tif"/>
</fig>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Performance of AMP/non-AMP classifiers with different algorithms on the test set.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Algorithm</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1-score</th>
<th align="center">Accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Random forest</td>
<td align="center">0.919</td>
<td align="center">0.917</td>
<td align="center">0.918</td>
<td align="center">0.918</td>
</tr>
<tr>
<td align="center">SVC</td>
<td align="center">0.945</td>
<td align="center">0.903</td>
<td align="center">0.923</td>
<td align="center">0.925</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">0.925</td>
<td align="center">0.901</td>
<td align="center">0.913</td>
<td align="center">0.914</td>
</tr>
<tr>
<td align="center">Na&#xef;ve Bayes</td>
<td align="center">0.845</td>
<td align="center">0.886</td>
<td align="center">0.865</td>
<td align="center">0.862</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-3">
<title>3.3 Feature selection</title>
<p>Here, we used the Propy Python library to extract linguistic and physicochemical-based properties and allowed cross-validation-based feature selection to select most important properties. The performance of models on the test set before feature selection is shown in <xref ref-type="sec" rid="s10">Supplementary Table S2</xref>. SVC (RBF) was the best classifier at this stage. Independent random forest models were trained on five independent sets from dataset 2&#x27;s train set, and features present in all final sets were selected (<xref ref-type="fig" rid="F2">Figure 2</xref>). The 1500&#x2b; properties were reduced to 51 features. The distribution of selected features among feature categories is shown in <xref ref-type="table" rid="T3">Table 3</xref>. Interestingly, all of these features were based on physicochemical properties of AMPs including net charge, molecular weight, charge density, aggregation propensity <italic>in vivo</italic>, distribution of charge and polarity along the peptide sequence, composition of non-polar residues, and buried residues. The relative importance of features from the random forest model is shown in <xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>. As can be seen in the figure, features have relatively similar importance. No linguistic-based property was found among selected features. A full list of selected features with their corresponding categories and relative importance is shown in <xref ref-type="sec" rid="s10">Supplementary Table S3</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Feature selection by the cross-validation process.</p>
</caption>
<graphic xlink:href="fmolb-10-1238509-g002.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Distribution of selected features in their categories.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Feature category</th>
<th align="left">Feature sub-category</th>
<th align="left">No.</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="7" align="left">Physicochemical</td>
<td align="left">Net charge</td>
<td align="right">1</td>
</tr>
<tr>
<td align="left">Molecular weight</td>
<td align="right">1</td>
</tr>
<tr>
<td align="left">Charge density</td>
<td align="right">1</td>
</tr>
<tr>
<td align="left">Aggregation propensity <italic>in vivo</italic>
</td>
<td align="right">1</td>
</tr>
<tr>
<td align="left">Physicochemical composition</td>
<td align="right">4</td>
</tr>
<tr>
<td align="left">Physicochemical distribution</td>
<td align="right">4</td>
</tr>
<tr>
<td align="left">Physicochemical transition</td>
<td align="right">1</td>
</tr>
<tr>
<td rowspan="3" align="left">Autocorrelation</td>
<td align="left">Geary autocorrelation</td>
<td align="right">5</td>
</tr>
<tr>
<td align="left">Moran autocorrelation</td>
<td align="right">10</td>
</tr>
<tr>
<td align="left">Normalized Moreau&#x2013;Broto autocorrelation</td>
<td align="right">6</td>
</tr>
<tr>
<td align="left">Pseudo-amino acid composition</td>
<td align="left">Pseudo-amino acid composition</td>
<td align="right">4</td>
</tr>
<tr>
<td rowspan="2" align="left">Sequence order</td>
<td align="left">Quasi-sequence order</td>
<td align="right">5</td>
</tr>
<tr>
<td align="left">Sequence-order coupling number</td>
<td align="right">8</td>
</tr>
<tr>
<td colspan="2" align="center">Total</td>
<td align="right">51</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-4">
<title>3.4 Performance of <italic>Staphylococcus aureus</italic>-specific activity classifier</title>
<p>After obtaining the final feature set, classification algorithms including random forest, SVC, linear SVC, KNN, and hybrid models were trained. Performances on the training set were compared using the ROC curves (<xref ref-type="fig" rid="F3">Figure 3</xref>). As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, the random forest model shows a better performance compared to linear SVC and KNN models.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>ROC curves obtained for different algorithms classifying AMPs active and non-active against <italic>S. aureus</italic>.</p>
</caption>
<graphic xlink:href="fmolb-10-1238509-g003.tif"/>
</fig>
<p>
<xref ref-type="table" rid="T4">Table 4</xref> shows the performance of models on the test sets. The Hamming distance between model performances shows that even models with very similar performances might get different results for the same AMP (<xref ref-type="sec" rid="s10">Supplementary Figure S2</xref>). Therefore, the hybrid model made by these classifiers shows potential to perform better on the test set. Several combinations of classifiers were also used to create hybrid classifiers, and performance on test results was obtained (<xref ref-type="sec" rid="s10">Supplementary Table S4</xref>). It can be seen that eclf5 (a combination of random forest, SVC polynomial, and linear SVC models) with an F1-score of 0.80 and a recall of 0.88 is the best performing classifier.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Performance of different algorithms in classifying AMPs active against <italic>Staphylococcus aureus</italic>.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Algorithm</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1-score</th>
<th align="center">Specificity</th>
<th align="center">Accuracy</th>
<th align="center">Balanced accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Random forest</td>
<td align="center">0.744</td>
<td align="center">0.859</td>
<td align="center">0.798</td>
<td align="center">0.711</td>
<td align="center">0.734</td>
<td align="center">0.785</td>
</tr>
<tr>
<td align="center">SVC polynomial</td>
<td align="center">0.739</td>
<td align="center">0.839</td>
<td align="center">0.786</td>
<td align="center">0.681</td>
<td align="center">0.721</td>
<td align="center">0.760</td>
</tr>
<tr>
<td align="center">KNN</td>
<td align="center">0.733</td>
<td align="center">0.821</td>
<td align="center">0.775</td>
<td align="center">0.656</td>
<td align="center">0.709</td>
<td align="center">0.739</td>
</tr>
<tr>
<td align="center">SVC RBF</td>
<td align="center">0.741</td>
<td align="center">0.777</td>
<td align="center">0.759</td>
<td align="center">0.624</td>
<td align="center">0.699</td>
<td align="center">0.700</td>
</tr>
<tr>
<td align="center">LSVC</td>
<td align="center">0.695</td>
<td align="center">0.857</td>
<td align="center">0.768</td>
<td align="center">0.650</td>
<td align="center">0.684</td>
<td align="center">0.754</td>
</tr>
<tr>
<td align="center">Hybrid</td>
<td align="center">0.7415</td>
<td align="center">0.8755</td>
<td align="center">0.8029</td>
<td align="center">0.729</td>
<td align="center">0.738</td>
<td align="center">0.802</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-5">
<title>3.5 Performance of the final shared model on the new independent set</title>
<p>The performance of the final constructed package of the hierarchical model was evaluated using a new dataset including AMPs and non-AMPs. Results in <xref ref-type="table" rid="T5">Table 5</xref> show that despite low precision, the model demonstrates high sensitivity and balanced accuracy in real-world scenarios.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Performance of the final hierarchical packaged model on an independent set.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1-score</th>
<th align="center">Specificity</th>
<th align="center">Accuracy</th>
<th align="center">Balanced accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">0.4792</td>
<td align="center">1.0000</td>
<td align="center">0.6479</td>
<td align="center">0.7340</td>
<td align="center">0.7863</td>
<td align="center">0.867</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>Growing AMP databases through gathering reported experimental and activity studies on AMPs has provided sufficient data for training many machine learning models for the prediction of AMP activity. Most prediction models concerning the antimicrobial activity of AMPs have been dedicated to distinguish AMPs from non-AMPs and were able to achieve high performances. However, as shown in the results of <xref ref-type="table" rid="T1">Table 1</xref>, the susceptibility to a single AMP is highly varied among different target species. Therefore, there is no guarantee that AMP candidates suggested by such predictive models will be able to show suitable activity against a specific species.</p>
<p>In the context of our study on detecting antimicrobial peptides active against <italic>S. aureus</italic>, we used hierarchical models to effectively tackle the complex nature of peptide space. We trained two separate classifiers. The first level of classification involved distinguishing between two broad categories: antimicrobial peptides (AMPs) and non-antimicrobial peptides (non-AMPs). Subsequently, we utilized the second level of classification to further analyze the subset of AMPs and distinguish between those that are active against <italic>S. aureus</italic> (SA-AMPs) and those that are not active against <italic>S. aureus</italic> (non-SA-AMPs) (<xref ref-type="fig" rid="F4">Figure 4</xref>).</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Schematic presentation of the hierarchical machine learning model pipeline to predict peptide antimicrobial activity.</p>
</caption>
<graphic xlink:href="fmolb-10-1238509-g004.tif"/>
</fig>
<p>Features extracted from peptides are based on either the amino acid letter sequence or physicochemical properties, and the performance of the final model on unseen peptides depends on the similarity of their properties with training data. However, only a very tiny portion of the sequence space is similar to the training set. Therefore, we expect classification models based on physicochemical properties of AMPs to generalize better. Notably, investigating all 51 features selected from 1,500&#x2b; features had a physicochemical nature, which is in agreement with our expectations.</p>
<p>The machine learning results demonstrate that the hierarchical model achieved promising performance in peptide classification. The first-level classifier showed an excellent F1-score (0.923) in differentiating between AMPs and non-AMPs, indicating its ability to capture key features that discriminate between these two categories. The second-level classifier, which focused specifically on classifying AMPs into SA-AMPs and non-SA-AMPs, also showed good performance with the F1-score (0.80) in distinguishing between these two categories.</p>
<p>The hierarchical approach provided several advantages over a single-classifier approach. By using a two-level classifier, we were able to first filter out non-AMPs and then further classify the remaining AMPs into two subcategories based on their activity against <italic>S. aureus</italic>. This hierarchical approach allows for using a more diverse library of peptides as the input compared to other studies, which have only used AMPs for training, and this is of particular interest in our research. Moreover, the hierarchical model is interpretable as it allows us to examine the performance and contributions of each level separately, making it easier to identify potential areas for improvement and fine-tuning.</p>
<p>Not many predictive tools are available to compare our results with theirs. <xref ref-type="bibr" rid="B14">Speck-Planche et al. (2016)</xref> and <xref ref-type="bibr" rid="B18">Vishnepolsky et al. (2022)</xref> made strain-specific predictive models to distinguish AMPs active against specific strains of bacteria, including <italic>S. aureus</italic>, and based on their raw performances, they are better classifiers. However, there are important considerations in practical applications and differences in the methodologies implemented. Both works only used AMPs as the training set. Therefore, their models are not familiar with non-AMP peptides. Considering this, if non-AMP peptides were to be used as input sequences, since they are not similar either to the positive or the negative set, it could result in undesirable performances. In the case of Vishnepolsky, B. et al. work, labeling AMPs was based on the MIC with the concentration unit of &#xb5;g/mL, which is not biologically reasonable since the transformation to &#xb5;M is necessary to allow the direct and accurate comparison of the inhibitory activities among the AMPs (<xref ref-type="bibr" rid="B14">Speck-Planche et al., 2016</xref>). In our work and many other recent works, &#xb5;g/mL concentrations are first converted to &#xb5;M (<xref ref-type="bibr" rid="B10">Kleandrova et al., 2016</xref>). These can limit the applicability of these models in practical applications.</p>
<p>It is worth noting that the performance of each level&#x2019;s classifier was evaluated using independent test sets, which were distinct from the training and validation sets utilized during model development. This approach ensured that the model&#x2019;s performance was generalizable and reflective of real-world scenarios. On the other hand, the final packaged hierarchical model also demonstrated strong performance on a mixed dataset of AMPs and non-AMPs, providing a significant advantage over other studies that were exclusively trained with AMPs. The results of our study suggest that the developed hierarchical model effectively classifies peptides into distinct categories, including distinguishing between SA-AMPs and non-SA-AMPs. This capability holds potential for various applications such as drug discovery, antimicrobial peptide design, and functional peptide annotation. The final package constructed here is publicly available on GitHub at: <ext-link ext-link-type="uri" xlink:href="https://github.com/h-khabaz/s.aureus-AMP-activity-calculator">https://github.com/h-khabaz/s.aureus-AMP-activity-calculator</ext-link>.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In conclusion, we developed a hierarchical machine learning model for peptide classification, specifically targeting the classification of antimicrobial peptides against <italic>S. aureus</italic>. The results demonstrate the effectiveness of the hierarchical approach in accurately classifying peptides into different categories and distinguishing between AMPs active and not-active against <italic>S. aureus</italic>. The developed model has potential applications in various fields, including drug discovery, peptide design, and functional annotation of peptides.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found at: <ext-link ext-link-type="uri" xlink:href="https://github.com/h-khabaz/s.aureus-amp-dataset">https://github.com/h-khabaz/s.aureus-amp-dataset</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>HK and AK conceived the idea. All authors involved in planning the project. MR-N and AK supervised the research. HK carried out the development part. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmolb.2023.1238509/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmolb.2023.1238509/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>D.-S.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Q.-S.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Y.-Z.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>propy: a tool to generate various modes of Chou&#x2019;s PseAAC</article-title>. <source>Bioinformatics</source> <volume>29</volume> (<issue>7</issue>), <fpage>960</fpage>&#x2013;<lpage>962</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt072</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chaudhary</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tuknait</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gautam</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mathur</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>A web server and mobile app for computing hemolytic potency of peptides</article-title>. <source>Sci. Rep.</source> <volume>6</volume>, <fpage>22843</fpage>. <pub-id pub-id-type="doi">10.1038/srep22843</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>T. K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Development and challenges of antimicrobial peptides for therapeutic applications</article-title>. <source>Antibiot. (Basel)</source> <volume>9</volume> (<issue>1</issue>), <fpage>24</fpage>. <pub-id pub-id-type="doi">10.3390/antibiotics9010024</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Eisner</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Poulin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Szafron</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Greiner</surname>
<given-names>R.</given-names>
</name>
</person-group> &#x201c;<article-title>Improving protein function prediction using the hierarchical structure of the gene ontology</article-title>,&#x201d; in <conf-name>Proceedings of the 2005 IEEE symposium on computational intelligence in bioinformatics and computational biology</conf-name>, <conf-loc>La Jolla, CA, USA</conf-loc>, <conf-date>November 2005</conf-date>, <fpage>1</fpage>&#x2013;<lpage>10</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gabere</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Noble</surname>
<given-names>W. S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Empirical comparison of web-based antimicrobial peptide prediction tools</article-title>. <source>Bioinformatics</source> <volume>33</volume> (<issue>13</issue>), <fpage>1921</fpage>&#x2013;<lpage>1929</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx081</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gogoladze</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Grigolava</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vishnepolsky</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chubinidze</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Duroux</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lefranc</surname>
<given-names>M.-P.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Dbaasp: database of antimicrobial activity and structure of peptides</article-title>. <source>FEMS Microbiol. Lett.</source> <volume>357</volume> (<issue>1</issue>), <fpage>63</fpage>&#x2013;<lpage>68</lpage>. <pub-id pub-id-type="doi">10.1111/1574-6968.12489</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hancock</surname>
<given-names>R. E.</given-names>
</name>
<name>
<surname>Sahl</surname>
<given-names>H.-G.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Antimicrobial and host-defense peptides as new anti-infective therapeutic strategies</article-title>. <source>Nat. Biotechnol.</source> <volume>24</volume> (<issue>12</issue>), <fpage>1551</fpage>&#x2013;<lpage>1557</lpage>. <pub-id pub-id-type="doi">10.1038/nbt1267</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>CD-HIT suite: A web server for clustering and comparing biological sequences</article-title>. <source>Bioinformatics</source> <volume>26</volume> (<issue>5</issue>), <fpage>680</fpage>&#x2013;<lpage>682</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq003</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ikuta</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Swetschinski</surname>
<given-names>L. R.</given-names>
</name>
<name>
<surname>Aguilar</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Sharara</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Mestrovic</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Gray</surname>
<given-names>A. P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Global mortality associated with 33 bacterial pathogens in 2019: A systematic analysis for the global burden of disease study 2019</article-title>. <source>Lancet</source> <volume>400</volume> (<issue>10369</issue>), <fpage>2221</fpage>&#x2013;<lpage>2248</lpage>. <pub-id pub-id-type="doi">10.1016/S0140-6736(22)02185-7</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kleandrova</surname>
<given-names>V. V.</given-names>
</name>
<name>
<surname>Ruso</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Speck-Planche</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dias Soeiro Cordeiro</surname>
<given-names>M. N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Enabling the discovery and virtual screening of potent and safe antimicrobial peptides. Simultaneous prediction of antibacterial activity and cytotoxicity</article-title>. <source>ACS Comb. Sci.</source> <volume>18</volume> (<issue>8</issue>), <fpage>490</fpage>&#x2013;<lpage>498</lpage>. <pub-id pub-id-type="doi">10.1021/acscombsci.6b00063</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>E. Y.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>M. W.</given-names>
</name>
<name>
<surname>Fulan</surname>
<given-names>B. M.</given-names>
</name>
<name>
<surname>Ferguson</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>G. C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>What can machine learning do for antimicrobial peptides, and what can antimicrobial peptides do for machine learning?</article-title> <source>Interface Focus</source> <volume>7</volume> (<issue>6</issue>), <fpage>20160153</fpage>. <pub-id pub-id-type="doi">10.1098/rsfs.2016.0153</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohamed</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Abdelkhalek</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Seleem</surname>
<given-names>M. N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Evaluation of short synthetic antimicrobial peptides for treatment of drug-resistant and intracellular <italic>Staphylococcus aureus</italic>
</article-title>. <source>Sci. Rep.</source> <volume>6</volume> (<issue>1</issue>), <fpage>29707</fpage>&#x2013;<lpage>29714</lpage>. <pub-id pub-id-type="doi">10.1038/srep29707</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedregosa</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Varoquaux</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gramfort</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Michel</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Thirion</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Grisel</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Scikit-learn: machine learning in Python</article-title>. <source>J. Mach. Learn. Res.</source> <volume>12</volume>, <fpage>2825</fpage>&#x2013;<lpage>2830</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1201.0490</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Speck-Planche</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kleandrova</surname>
<given-names>V. V.</given-names>
</name>
<name>
<surname>Ruso</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Ds Cordeiro</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>First multitarget chemo-bioinformatic model to enable the discovery of antibacterial peptides against multiple Gram-positive pathogens</article-title>. <source>J. Chem. Inf. Model.</source> <volume>56</volume> (<issue>3</issue>), <fpage>588</fpage>&#x2013;<lpage>598</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.5b00630</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stryjewski</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Chambers</surname>
<given-names>H. F.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Skin and soft-tissue infections caused by community-acquired methicillin-resistant <italic>Staphylococcus aureus</italic>
</article-title>. <source>Clin. Infect. Dis.</source> <volume>46</volume>, <fpage>S368</fpage>&#x2013;<lpage>S377</lpage>. <pub-id pub-id-type="doi">10.1086/533593</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Valentini</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>True path rule hierarchical ensembles for genome-wide gene function prediction</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>8</volume> (<issue>3</issue>), <fpage>832</fpage>&#x2013;<lpage>847</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2010.38</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vishnepolsky</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gabrielian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rosenthal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hurt</surname>
<given-names>D. E.</given-names>
</name>
<name>
<surname>Tartakovsky</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Managadze</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Predictive model of linear antimicrobial peptides active against gram-negative bacteria</article-title>. <source>J. Chem. Inf. Model.</source> <volume>58</volume> (<issue>5</issue>), <fpage>1141</fpage>&#x2013;<lpage>1151</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.8b00118</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vishnepolsky</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Grigolava</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Managadze</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gabrielian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rosenthal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hurt</surname>
<given-names>D. E.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Comparative analysis of machine learning algorithms on the microbial strain-specific AMP prediction</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume> (<issue>4</issue>), <fpage>bbac233</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac233</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vishnepolsky</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Pirtskhalava</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Prediction of linear cationic antimicrobial peptides based on characteristics responsible for their interaction with the membranes</article-title>. <source>J. Chem. Inf. Model</source> <volume>54</volume> (<issue>5</issue>), <fpage>1512</fpage>&#x2013;<lpage>1523</lpage>. <pub-id pub-id-type="doi">10.1021/ci4007003</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wehrmann</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cerri</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Barros</surname>
<given-names>R.</given-names>
</name>
</person-group> &#x201c;<article-title>Hierarchical multi-label classification networks</article-title>,&#x201d; in <conf-name>Proceedings of the International conference on machine learning: PMLR</conf-name>, <conf-loc>Stockholmsm&#xe4;ssan, Stockholm, Sweden</conf-loc>, <conf-date>July 2018</conf-date>, <fpage>5075</fpage>&#x2013;<lpage>5084</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wolcott</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rhoads</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bennett</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wolcott</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gogokhia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Costerton</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Chronic wounds and the medical biofilm paradigm</article-title>. <source>J. wound care</source> <volume>19</volume> (<issue>2</issue>), <fpage>45</fpage>&#x2013;<lpage>46</lpage>. <pub-id pub-id-type="doi">10.12968/jowc.2010.19.2.46966</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wolfram Research</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Mathematica</source>. <publisher-loc>Champaign, Illinois, USA</publisher-loc>: <publisher-name>Wolfram Research</publisher-name>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zasloff</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Antimicrobial peptides of multicellular organisms</article-title>. <source>nature</source> <volume>415</volume> (<issue>6870</issue>), <fpage>389</fpage>&#x2013;<lpage>395</lpage>. <pub-id pub-id-type="doi">10.1038/415389a</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>