<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1211020</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2023.1211020</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Recognition of outer membrane proteins using multiple feature fusion</article-title>
<alt-title alt-title-type="left-running-head">Su et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2023.1211020">10.3389/fgene.2023.1211020</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Su</surname>
<given-names>Wenxia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2211299/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Qian</surname>
<given-names>Xiaojun</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2332036/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Keli</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2220408/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ding</surname>
<given-names>Hui</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/190127/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Huang</surname>
<given-names>Chengbing</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2330695/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhang</surname>
<given-names>Zhaoyue</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1503844/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Science</institution>, Inner Mongolia Agriculture University, <addr-line>Hohhot</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Life Science and Technology</institution>, <institution>Center for Information Biology</institution>, University of Electronic Science and Technology of China, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Nonlinear Research Institute</institution>, Baoji University of Arts and Sciences, <addr-line>Baoji</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>School of Computer Science and Technology</institution>, Aba Teachers University, <addr-line>Aba</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>School of Healthcare Technology</institution>, Chengdu Neusoft University, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/827295/overview">Lei Chen</ext-link>, Shanghai Maritime University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/855926/overview">Yongqiang Xing</ext-link>, Inner Mongolia University of Science and Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/567032/overview">Cangzhi Jia</ext-link>, Dalian Maritime University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Chengbing Huang, <email>abtchcb@qq.com</email>; Zhaoyue Zhang, <email>zyzhang@uestc.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>06</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1211020</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>04</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>05</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Su, Qian, Yang, Ding, Huang and Zhang.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Su, Qian, Yang, Ding, Huang and Zhang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Introduction:</bold> Outer membrane proteins are crucial in maintaining the structural stability and permeability of the outer membrane. Outer membrane proteins exhibit several functions such as antigenicity and strong immunogenicity, which have potential applications in clinical diagnosis and disease prevention. However, wet experiments for studying OMPs are time and capital-intensive, thereby necessitating the use of computational methods for their identification.</p>
<p>
<bold>Methods:</bold> In this study, we developed a computational model to predict outer membrane proteins. The non-redundant dataset consists of a positive set of 208 outer membrane proteins and a negative set of 876 non-outer membrane proteins. In this study, we employed the pseudo amino acid composition method to extract feature vectors and subsequently utilized the support vector machine for prediction.</p>
<p>
<bold>Results and Discussion:</bold> In the Jackknife cross-validation, the overall accuracy and the area under receiver operating characteristic curve were observed to be 93.19% and 0.966, respectively. These results demonstrate that our model can produce accurate predictions, and could serve as a valuable guide for experimental research on outer membrane proteins.</p>
</abstract>
<kwd-group>
<kwd>outer membrane protein</kwd>
<kwd>pseudo amino acid composition</kwd>
<kwd>support vector machine</kwd>
<kwd>jackknife test</kwd>
<kwd>prediction model</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>research-article</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Outer membrane proteins (OMPs) are a special type of proteins that are found in the outermost membranes of Gram-negative bacteria, mitochondria, and chloroplasts (<xref ref-type="bibr" rid="B30">Rollauer et al., 2015</xref>; <xref ref-type="bibr" rid="B29">Qi et al., 2022</xref>). OMPs serve a wide range of functions, including acting as adhesion factors in virulence, channels for small hydrophilic molecules, enzymes in biochemical reactions, and antigens in immune responses. They also work in concert with other substances to enhance the bacteria pathogenicity. Recent research on OMPs has revealed their potential for clinical diagnosis and disease prevention. Several published studies have explored OMPs as potential vaccine candidates (<xref ref-type="bibr" rid="B4">Budiardjo et al., 2021</xref>; <xref ref-type="bibr" rid="B11">Fahie et al., 2021</xref>; <xref ref-type="bibr" rid="B7">Cheng et al., 2022</xref>; <xref ref-type="bibr" rid="B46">Yu et al., 2022</xref>). The functions are determined by the OMP&#x2019;s structure and the way it interacts with other molecules. OMPs are typically composed of a transmembrane &#x3b2;-barrel architecture, providing permeability to the outer membrane and maintaining structural stability. Among different types of OMPs, &#x3b2;-buckets consist of varying even numbers of &#x3b2;-folding sheets, ranging from 8 to 26 (<xref ref-type="bibr" rid="B30">Rollauer et al., 2015</xref>). The specific composition of the &#x3b2;-barrel architecture is determined by the amino acid sequence of the OMPs. Mutations in sequences can impact the stability and function of the protein.</p>
<p>Distinguishing OMPs from non-OMPs can aid researchers in identifying promising vaccine targets, developing new antibiotics and therapeutics, and understanding the evolution of Gram-negative bacteria. Despite their distinctive &#x3b2;-barrel structure, OMPs are exposed to numerous charged and polar residues in the membrane, making it challenging to distinguish them from non-OMPs. This is a primary challenge and a significant obstacle in the research process, given the considerable time and capital costs associated with laboratory studies of OMPs. As a result, OMP prediction has tremendous significance for the scientific community. Currently, various machine learning methods have been used for the identification of OMPs, such as support vector machine (SVM) (<xref ref-type="bibr" rid="B28">Park et al., 2005</xref>; <xref ref-type="bibr" rid="B18">Gromiha et al., 2006</xref>; <xref ref-type="bibr" rid="B21">Hu et al., 2017</xref>; <xref ref-type="bibr" rid="B47">Zhang et al., 2021</xref>), k-nearest neighbor (K-NN) method (<xref ref-type="bibr" rid="B43">Yan et al., 2008</xref>), neural networks (NN) (<xref ref-type="bibr" rid="B21">Hu et al., 2017</xref>). These methods utilize the amino acid composition, and physical and chemical properties of the amino acid sequences to construct the prediction models. <xref ref-type="bibr" rid="B17">Gromiha and Suwa (2003)</xref>; <xref ref-type="bibr" rid="B15">Gromiha and Suwa (2005)</xref> developed multiple OMP prediction methods based on amino acid composition, residue pair preference, and motif sequence. However, these methods only achieved prediction accuracies of 80%&#x2013;90%. Subsequently, a machine learning algorithm was proposed with a higher accuracy ranging from 90% to 94% (<xref ref-type="bibr" rid="B14">Gromiha et al., 2005</xref>; <xref ref-type="bibr" rid="B18">Gromiha et al., 2006</xref>). <xref ref-type="bibr" rid="B25">Lin (2008)</xref> further improved the OMP prediction model by introducing the Incremental Diversity with Quality Distinctness analysis, which combines the Markov discriminant method and the pseudo amino acid composition (Pse-AAC). Despite the progress made in OMP predictions, there is still room for further improvement in prediction quality.</p>
<p>In this article, we proposed a novel method for predicting OMPs that combines Pse-AAC and SVM. To extract the features for amino acid composition and physical and chemical characteristics of amino acids, we used the Pse-AAC feature extraction method. Additionally, we introduced multi-level amino acid residue index correlation coefficients such as hydrophobic value, average polarity, and solvation-free energy to enhance the accuracy of our prediction model. To assess the effectiveness and reliability of our approach, we also conducted a comprehensive comparison and analysis of our proposed model with existing methods for predicting OMPs. Our developed approach will be useful for distinguishing OMPs from non-OMPs.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Datasets</title>
<p>The construction of a reliable dataset is the basis for developing an accurate outer membrane protein prediction model (<xref ref-type="bibr" rid="B35">Su et al., 2021</xref>). A well-designed dataset is crucial for developing effective algorithms and an objective evaluation and prediction system. In this paper, membrane proteins were extracted from the PSORT-B database (<ext-link ext-link-type="uri" xlink:href="https://www.psort.org/">https://www.psort.org/</ext-link>) (<xref ref-type="bibr" rid="B13">Gardy et al., 2003</xref>), and globular proteins were extracted from the PDB40D of SCOP_1.37 database (<ext-link ext-link-type="uri" xlink:href="http://scop.mrc-lmb.cam.ac.uk/scop/">http://scop.mrc-lmb.cam.ac.uk/scop/</ext-link>) (<xref ref-type="bibr" rid="B1">Andreeva et al., 2020</xref>). As a result, a total of 208 OMPs were selected as the positive set, while 879 non-OMPs were chosen as the negative set. The negative set included 206 inner membrane proteins and 673 globular proteins. The globular protein dataset contained 154 complete <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> proteins, 156 complete <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> proteins, 184 <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> proteins, and 179 <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> proteins. Since the sequence homology of each protein class was less than 40%, proteins in each database were not similar and were de-redundant.</p>
</sec>
<sec id="s2-2">
<title>2.2 Feature encoding</title>
<p>To construct a prediction model, it is necessary to represent the protein sequences as mathematical vectors. This conversion is commonly known as feature extraction (<xref ref-type="bibr" rid="B3">Basith et al., 2020</xref>; <xref ref-type="bibr" rid="B10">Dao et al., 2022b</xref>; <xref ref-type="bibr" rid="B50">Zhang Z.-Y. et al., 2022</xref>; <xref ref-type="bibr" rid="B22">Hunt et al., 2022</xref>; <xref ref-type="bibr" rid="B24">Karuna Nidhi et al., 2022</xref>; <xref ref-type="bibr" rid="B36">Sun et al., 2022</xref>; <xref ref-type="bibr" rid="B38">Tran and Nguyen, 2022</xref>; <xref ref-type="bibr" rid="B40">Wang et al., 2022</xref>; <xref ref-type="bibr" rid="B45">Yang et al., 2022</xref>). The amino acid composition (ACC) of the protein has a great impact on protein classification research (<xref ref-type="bibr" rid="B2">Awais et al., 2021</xref>; <xref ref-type="bibr" rid="B33">Shoombuatong et al., 2022b</xref>; <xref ref-type="bibr" rid="B27">Manavalan and Patra, 2022</xref>; <xref ref-type="bibr" rid="B31">Rout et al., 2022</xref>; <xref ref-type="bibr" rid="B52">Zhu et al., 2022</xref>). By using the ACC, a protein sequence can be represented as a 20-D (dimension) vector as follows:<disp-formula id="e1">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mn>20</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>In Eq. <xref ref-type="disp-formula" rid="e1">1</xref>, <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mo>&#x2211;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represented the number of the <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mn>20</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> amino acid in the protein sequence.</p>
<p>The type of amino acids is determined by their side chains, as the 20 types of amino acid side chains differ in shape, size, negativity, hydrophobicity, and acid-base properties. The distinct characteristics of the 20 amino acid side chains result in various combinations of amino acid sequences that exhibit different structures and functions. Therefore, algorithms based on the physicochemical properties of amino acids are another major category of feature extraction methods. Pse-AAC, originally proposed by Chou, is a feature extraction algorithm, that is, based on the physical and chemical properties of amino acids (<xref ref-type="bibr" rid="B8">Chou, 2005</xref>). By using Pse-AAC, a protein sample can be represented as follows:<disp-formula id="e2">
<mml:math id="m9">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>20</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mn>20</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mn>20</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where the first 20 numbers in Eq. <xref ref-type="disp-formula" rid="e2">2</xref> are the classic AAC features, and the next &#x3bb; discrete numbers represent the position information of residues in amino acid sequences. For different problems, the optimal value of <inline-formula id="inf8">
<mml:math id="m10">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> may vary. In this study, we selected the optimal value of &#x3bb; that yielded the highest sensitivity through the jackknife test.</p>
</sec>
<sec id="s2-3">
<title>2.3 Support vector machine</title>
<p>SVM is a powerful supervised machine learning classification method based on statistical learning theory (<xref ref-type="bibr" rid="B26">Manavalan et al., 2019</xref>). It was originally designed based on the idea of the generalized linear classifier. First, features were mapped to high-dimensional space. Next, a separating hyperplane is constructed to separate the two categories in the high-dimensional feature space (<xref ref-type="bibr" rid="B39">Vapnik and Control, 2019</xref>). To avoid expensive computations, the mapping function only involves the relatively low-dimensional vector in the input space and the dot product in the feature space. The global optimization approach and avoidance of overfitting in SVM have made it a successful tool for addressing various bioinformatics problems (<xref ref-type="bibr" rid="B48">Zhang H. et al., 2022</xref>). In this paper, the support vector machine (SVM) was implemented using the widely used software LIBSVM (<ext-link ext-link-type="uri" xlink:href="http://www.csie.ntu.edu.tw/%7Ecjlin/libsvm">http://www.csie.ntu.edu.tw/&#x223c;cjlin/libsvm</ext-link>) (<xref ref-type="bibr" rid="B6">Chang and Lin, 2011</xref>). The radial basis function which is defined as <inline-formula id="inf9">
<mml:math id="m11">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> was chosen as the kernel function. The regularization parameter <italic>C</italic> and the kernel width parameter <italic>&#x3b3;</italic> were optimized on the training set using a grid search strategy.</p>
</sec>
<sec id="s2-4">
<title>2.4 Evaluation methods</title>
<p>At present, k-fold cross-validation and jackknife cross-validation are widely used for prediction evaluation (<xref ref-type="bibr" rid="B37">Tabaie et al., 2021</xref>; <xref ref-type="bibr" rid="B9">Dao et al., 2022a</xref>; <xref ref-type="bibr" rid="B42">Xiao et al., 2022</xref>; <xref ref-type="bibr" rid="B51">Zhou et al., 2022</xref>). The jackknife test is a type of cross-validation that involves leaving one observation out of the dataset at a time and using the remaining observations to train a model. This process is repeated for each observation in the dataset, resulting in n different models, where n is the number of observations in the dataset. In this article, we used the Jackknife test to evaluate the prediction results. The sensitivity (<italic>S</italic>
<sub>
<italic>n</italic>
</sub>), specificity (<italic>S</italic>
<sub>
<italic>p</italic>
</sub>), average accuracy (<italic>AA</italic>), overall prediction accuracy (<italic>OA</italic>), and Matthew&#x2019;s correlation coefficient (<italic>MCC</italic>), the area under ROC curve (auROC) were used to evaluate the prediction performance of the algorithm (<xref ref-type="bibr" rid="B44">Yang et al., 2021</xref>; <xref ref-type="bibr" rid="B49">Zhang Q. et al., 2022</xref>). The evaluation metrics are defined as follows:<disp-formula id="e3">
<mml:math id="m12">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m13">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m14">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m15">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>A</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m16">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <italic>TP</italic> represents the number of the positive sample correctly identified, <italic>FN</italic> represents the positive sample wrongly identified as a negative sample, <italic>FP</italic> represents the negative sample wrongly identified as a positive sample, and <italic>TN</italic> represents the negative sample correctly identified. AuROC is an indicator that relates to the receiver operating characteristic (ROC) curve, which is a plot of a series of continuous (1-<italic>Sp</italic>) values on the horizontal axis against their corresponding Sn values on the vertical axis. The ROC curve is a useful tool for evaluating the sensitivity and specificity of a model (<xref ref-type="bibr" rid="B20">Hasan et al., 2022</xref>; <xref ref-type="bibr" rid="B23">Jeon et al., 2022</xref>). AuROC is calculated in this study as an indicator of classification ability and performance. A larger auROC value indicates better performance and classification ability of the model.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<title>3 Results and discussion</title>
<sec id="s3-1">
<title>3.1 Model performance</title>
<p>In this study, the proteins were first obtained in FASTA format and then the PseAAC program (<xref ref-type="bibr" rid="B32">Shen and Chou, 2008</xref>) was used to extract the feature vectors of pseudo amino acid components. To achieve relatively optimal prediction results, different parameters were selected to extract pseudo amino acid component feature vectors of protein sequences. Specifically, feature vectors were extracted using different values of <inline-formula id="inf10">
<mml:math id="m17">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (the weight factor) including 0.1, 0.2, 0.3, 0.4, 0.5, and 0.6, and <inline-formula id="inf11">
<mml:math id="m18">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> was taken as either 3 or 5. The extracted feature vectors were then used for prediction using different values of <inline-formula id="inf12">
<mml:math id="m19">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> including 0.04, 0.05, 0.06, 0.07, 0.08, and 0.09. The SVM model was trained using svm-train in LIBSVM, and the optimal parameter array and optimal feature subset were searched from the prediction results. Only the <inline-formula id="inf13">
<mml:math id="m20">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> value that achieved the optimal prediction result was selected and listed in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The performance comparison of prediction models under different parameter conditions.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">
<inline-formula id="inf14">
<mml:math id="m21">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf15">
<mml:math id="m22">
<mml:mrow>
<mml:mi mathvariant="bold">&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf16">
<mml:math id="m23">
<mml:mrow>
<mml:mi mathvariant="bold">&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">Sn(%)</th>
<th align="center">
<italic>Sp</italic>(%)</th>
<th align="center">
<italic>MCC</italic>(%)</th>
<th align="center">
<italic>OA</italic> (%)</th>
<th align="center">
<italic>AA</italic> (%)</th>
<th align="center">AuROC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">0.1,3,0.05</td>
<td align="left">78.37</td>
<td align="left">96.59</td>
<td align="left">77.16</td>
<td align="left">93.10</td>
<td align="left">87.48</td>
<td align="left">0.962</td>
</tr>
<tr>
<td align="left">0.2,3,0.08</td>
<td align="left">78.85</td>
<td align="left">96.59</td>
<td align="left">77.49</td>
<td align="left">93.19</td>
<td align="left">87.72</td>
<td align="left">0.966</td>
</tr>
<tr>
<td align="left">0.3,3,0.09</td>
<td align="left">75.96</td>
<td align="left">96.59</td>
<td align="left">75.46</td>
<td align="left">92.64</td>
<td align="left">86.27</td>
<td align="left">0.965</td>
</tr>
<tr>
<td align="left">0.4,3,0.08</td>
<td align="left">79.33</td>
<td align="left">95.79</td>
<td align="left">75.97</td>
<td align="left">92.64</td>
<td align="left">87.56</td>
<td align="left">0.962</td>
</tr>
<tr>
<td align="left">0.5,3,0.09</td>
<td align="left">79.33</td>
<td align="left">95.79</td>
<td align="left">75.97</td>
<td align="left">92.64</td>
<td align="left">87.56</td>
<td align="left">0.961</td>
</tr>
<tr>
<td align="left">0.6,3,0.09</td>
<td align="left">79.81</td>
<td align="left">95.90</td>
<td align="left">76.57</td>
<td align="left">92.82</td>
<td align="left">87.86</td>
<td align="left">0.958</td>
</tr>
<tr>
<td align="left">0.1,5,0.07</td>
<td align="left">79.81</td>
<td align="left">96.59</td>
<td align="left">78.17</td>
<td align="left">93.38</td>
<td align="left">88.20</td>
<td align="left">0.965</td>
</tr>
<tr>
<td align="left">0.2,5,0.09</td>
<td align="left">79.81</td>
<td align="left">95.56</td>
<td align="left">75.79</td>
<td align="left">92.55</td>
<td align="left">87.69</td>
<td align="left">0.968</td>
</tr>
<tr>
<td align="left">0.3,5,0.09</td>
<td align="left">80.77</td>
<td align="left">95.45</td>
<td align="left">76.22</td>
<td align="left">92.64</td>
<td align="left">88.11</td>
<td align="left">0.966</td>
</tr>
<tr>
<td align="left">0.4,5,0.08</td>
<td align="left">81.73</td>
<td align="left">95.11</td>
<td align="left">76.15</td>
<td align="left">92.55</td>
<td align="left">88.42</td>
<td align="left">0.964</td>
</tr>
<tr>
<td align="left">0.5,5,0.07</td>
<td align="left">84.61</td>
<td align="left">95.11</td>
<td align="left">78.19</td>
<td align="left">93.10</td>
<td align="left">89.86</td>
<td align="left">0.963</td>
</tr>
<tr>
<td align="left">0.6,5,0.07</td>
<td align="left">81.25</td>
<td align="left">94.43</td>
<td align="left">74.34</td>
<td align="left">91.90</td>
<td align="left">87.84</td>
<td align="left">0.956</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In this study, the benchmark dataset consisted of 208 OMPs and 879 non-OMPs. Due to this imbalanced dataset, using average accuracy as the sole evaluation criterion may lead to skewed results toward the negative sets. Thus, the paper used overall accuracy as the main criterion for model evaluation. By analyzing the data in <xref ref-type="table" rid="T1">Table 1</xref>, it was observed that high prediction sensitivity was achieved using Jackknife cross-validation with different parameters. And, with the best prediction result obtained with a weight factor of 0.5, the parameter <inline-formula id="inf17">
<mml:math id="m24">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> taking 5, <inline-formula id="inf18">
<mml:math id="m25">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> taking 0.07, resulting in an overall accuracy of 93.10%.</p>
</sec>
<sec id="s3-2">
<title>3.2 Model comparison</title>
<p>Various methods have been proposed by different researchers to predict and distinguish OMPs from other types of membrane proteins. <xref ref-type="bibr" rid="B41">Wu et al. (2007)</xref> proposed a prediction method that uses information differences to compare the distribution of subsequences and residual sequences, resulting in a prediction accuracy of 99.20%. <xref ref-type="bibr" rid="B43">Yan et al. (2008)</xref> proposed a method based on the K-nearest neighbor (KNN) method, which predicted the weighted Euclidean distance calculated by residual synthesis and achieved a recognition accuracy of 96.1%, sensitivity of 87.5%, specificity of 98.2% with 0.873 MCC. <xref ref-type="bibr" rid="B18">Gromiha et al. (2006)</xref> discriminate of OMPs and non-OMPs using different machine learning approaches, the best performance achieved sensitivity of 84.6%, specificity of 95.8% and accuracy of 93.7%. And the SVM-based model achieved sensitivity of 72.6%, specificity of 98.2% and accuracy of 93.3%. <xref ref-type="bibr" rid="B28">Park et al. (2005)</xref> proposed an SVM method that considers both amino acid composition and residue pair information, achieving sensitivity of 90.9 %, specificity of 94.7%, MCC 0.816 of and accuracy of 93.9%. <xref ref-type="bibr" rid="B12">Gao et al. (2010)</xref> developed a method that combined the structural and physicochemical characteristics of sequence-derived proteins with amino acid composition to distinguish OMPs and non-OMPs using SVM, with an overall accuracy of 97.8%, sensitivity of 91.8 %, specificity of 99.2% and MCC 0.928.</p>
<p>In this paper, the model constructed using the SVM algorithm achieved an overall accuracy of 93.10% and auROC of 0.963 under Jackknife cross-validation, respectively. Besides, the sensitivity, specificity, MCC, and average accuracy were found to be 84.61%, 95.11%, 78.19%, and 89.86%, respectively. Compared to previous SVM-based models, some progress has been made.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>This article focused on the prediction and recognition of OMPs using the method of combining Pse-AAC with SVM. The study achieved good results with the Pse-AAC method, which not only considers the content of 20 natural amino acids in each protein sequence but also includes the correlation between various amino acids, such as physical and chemical properties. This approach is more advanced than traditional methods that only consider amino acid composition, leading to more accurate prediction results. SVM is a widely used algorithm in bioinformatics (<xref ref-type="bibr" rid="B19">Hasan et al., 2020</xref>; <xref ref-type="bibr" rid="B34">Shoombuatong et al., 2022a</xref>; <xref ref-type="bibr" rid="B5">Bupi et al., 2023</xref>), and applying it to the prediction of OMPs is an inevitable trend in current research. The constructed model using the SVM algorithm achieved high performance with an overall accuracy of 93.10% and auROC of 0.963 under Jackknife cross-validation. The sensitivity, specificity, Matthew correlation coefficient, and average accuracy achieved 84.61%, 95.11%, 78.19%, and 89.86%, respectively. However, while feature extraction algorithms have been widely used in prediction methods and have achieved good performance, the relationship between the extracted information and protein structure and function needs to be further explored. This challenge will undoubtedly be the focus of our future research efforts aimed at identifying OMPs. The development of accurate prediction models for OMPs has the potential to significantly impact fields ranging from antibiotic discovery and vaccine development to biotechnology and bacterial diagnostics.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>Conceptualization, HD and ZZ; data curation, WS and XQ; formal analysis, WS, XQ, and KY; funding acquisition, WS and ZZ; supervision, CH; writing&#x2014;original draft, WS; writing&#x2014;review and editing, CH and ZZ. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This research was funded by grants from the National Natural Science Foundation of China (Grant Nos. 62201299, 62102067), Natural Science Foundation of the Inner Mongolia of China (Grant No. 2021BS06003), Science and Technology Research Project of Colleges and Universities in Inner Mongolia of China (Grant No. NJZY21473), and Basic Scientific Research Foundation of Colleges and Universities directly under Inner Mongolia of China (Grant No. BR220505).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andreeva</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kulesha</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Gough</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Murzin</surname>
<given-names>A. G. J. N. A. R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The SCOP database in 2020: Expanded classification of representative family and superfamily domains of known protein structures</article-title>. <source>Nucleic Acids Res.</source> <volume>48</volume>, <fpage>D376</fpage>&#x2013;<lpage>D382</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz1064</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Awais</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hussain</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Rasool</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>Y. D. J. C. B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>iTSP-PseAAC: identifying tumor suppressor proteins by using fully connected neural network and PseAAC</article-title>. <source>Curr. Bioinform.</source> <volume>16</volume>, <fpage>700</fpage>&#x2013;<lpage>709</lpage>. <pub-id pub-id-type="doi">10.2174/15748936mtezfmteby</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Basith</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Manavalan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hwan Shin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Machine intelligence in peptide therapeutics: A next-generation tool for rapid disease screening</article-title>. <source>Med. Res. Rev.</source> <volume>40</volume>, <fpage>1276</fpage>&#x2013;<lpage>1314</lpage>. <pub-id pub-id-type="doi">10.1002/med.21658</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Budiardjo</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Ikujuni</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Firlar</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Cordova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kaelber</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Slusky</surname>
<given-names>J. S. J. T. J. O. M. B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>High-yield preparation of outer membrane protein efflux pumps by <italic>in vitro</italic> refolding is concentration dependent</article-title>. <source>J. Membr. Biol.</source> <volume>254</volume>, <fpage>41</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1007/s00232-020-00161-y</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bupi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Sangaraju</surname>
<given-names>V. K.</given-names>
</name>
<name>
<surname>Phan</surname>
<given-names>L. T.</given-names>
</name>
<name>
<surname>Lal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vo</surname>
<given-names>T. T. B.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>P. T.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>An effective integrated machine learning framework for identifying severity of tomato yellow leaf curl virus and their experimental validation</article-title>. <source>Research</source> <volume>6</volume>, <fpage>0016</fpage>. <pub-id pub-id-type="doi">10.34133/research.0016</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>C. J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Libsvm: A library for support vector machines</article-title>. <source>ACM Trans. Intelligent Syst. Technol.</source> <volume>2</volume>, <fpage>1</fpage>&#x2013;<lpage>27</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>gutMGene: a comprehensive database for target genes of gut microbes and microbial metabolites</article-title>. <source>Nucleic Acids Res.</source> <volume>50</volume>, <fpage>D795</fpage>&#x2013;<lpage>D800</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab786</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chou</surname>
<given-names>K.-C. J. B.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Using amphiphilic pseudo amino acid composition to predict enzyme subfamily classes</article-title>. <source>Bioinformatics</source> <volume>21</volume>, <fpage>10</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bth466</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dao</surname>
<given-names>F.-Y.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Fullwood</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H. J. R.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Accurate identification of DNA replication origin by fusing epigenomics and chromatin interaction information</article-title>. <source>Res. (Wash D C)</source> <volume>2022</volume>, <fpage>9780293</fpage>. <pub-id pub-id-type="doi">10.34133/2022/9780293</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dao</surname>
<given-names>F.-Y.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.-Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H. J. C. B.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>BDselect: A package for k-mer selection based on the binomial distribution</article-title>. <source>Curr. Bioinforma</source>. <volume>17</volume>, <fpage>238</fpage>&#x2013;<lpage>244</lpage>. <pub-id pub-id-type="doi">10.2174/1574893616666211007102747</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fahie</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chisholm</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M. J. N. T. M.</given-names>
</name>
<name>
<surname>Protocols</surname>
</name>
</person-group> (<year>2021</year>). <article-title>Protein analyte sensing with an outer membrane protein G (OmpG) nanopore</article-title>. <source>Methods Mol. Biol.</source> <volume>186</volume>, <fpage>77</fpage>&#x2013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-0716-0806-7_7</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>Q.-B.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>X.-F.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>Z.-C.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>J. J. A. B.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Improving discrimination of outer membrane proteins by fusing different forms of pseudo amino acid composition</article-title>. <source>Anal. Biochem.</source> <volume>398</volume>, <fpage>52</fpage>&#x2013;<lpage>59</lpage>. <pub-id pub-id-type="doi">10.1016/j.ab.2009.10.040</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gardy</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Spencer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ester</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tusnady</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Simon</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2003</year>). <article-title>PSORT-B: Improving protein subcellular localization prediction for Gram-negative bacteria</article-title>. <source>Nucleic Acids Res.</source> <volume>31</volume>, <fpage>3613</fpage>&#x2013;<lpage>3617</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkg602</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gromiha</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Ahmad</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Suwa</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Application of residue distribution along the sequence for discriminating outer membrane proteins</article-title>. <source>Comput. Biol. Chem.</source> <volume>29</volume>, <fpage>135</fpage>&#x2013;<lpage>142</lpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2005.02.006</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gromiha</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Suwa</surname>
<given-names>M. J. B.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>A simple statistical method for discriminating outer membrane proteins with better accuracy</article-title>. <source>Bioinformatics</source> <volume>21</volume>, <fpage>961</fpage>&#x2013;<lpage>968</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bti126</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gromiha</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Suwa</surname>
<given-names>M. J. B. E. B. A.-P.</given-names>
</name>
<name>
<surname>Proteomics</surname>
</name>
</person-group> (<year>2006</year>). <article-title>Influence of amino acid properties for discriminating outer membrane proteins at better accuracy</article-title>. <source>Biochim. Biophys. Acta</source> <volume>1764</volume>, <fpage>1493</fpage>&#x2013;<lpage>1497</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbapap.2006.07.005</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gromiha</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Suwa</surname>
<given-names>M. J. I. J. O. B. M.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Variation of amino acid properties in all-&#x3b2; globular and outer membrane protein structures</article-title>. <source>Int. J. Biol. Macromol.</source> <volume>32</volume>, <fpage>93</fpage>&#x2013;<lpage>98</lpage>. <pub-id pub-id-type="doi">10.1016/s0141-8130(03)00042-4</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gromiha</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Suwa</surname>
<given-names>M. J. P. S.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Discrimination of outer membrane proteins using machine learning algorithms</article-title>. <source>Funct. Bioinforma. Proteins</source> <volume>63</volume>, <fpage>1031</fpage>&#x2013;<lpage>1037</lpage>. <pub-id pub-id-type="doi">10.1002/prot.20929</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hasan</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Schaduangrat</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Basith</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Shoombuatong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Manavalan</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>HLPpred-fuse: Improved and robust prediction of hemolytic peptide and its activity by fusing multiple feature representation</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>3350</fpage>&#x2013;<lpage>3356</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa160</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hasan</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Tsukiyama</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>J. Y.</given-names>
</name>
<name>
<surname>Kurata</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Alam</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Deepm5C: A deep-learning-based hybrid framework for identifying human rna N5-methylcytosine sites using a stacking strategy</article-title>. <source>Mol. Ther.</source> <volume>30</volume>, <fpage>2856</fpage>&#x2013;<lpage>2867</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymthe.2022.05.001</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ju</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Measuring disease similarity and predicting disease-related ncRNAs by a novel method</article-title>. <source>Bmc Med. Genomics</source> <volume>10</volume>, <fpage>71</fpage>. <pub-id pub-id-type="doi">10.1186/s12920-017-0315-9</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hunt</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Montgomery</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Berkenpas</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Sigafoos</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Oakley</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Espinosa</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Recent progress of machine learning in gene therapy</article-title>. <source>Curr. Gene Ther.</source> <volume>22</volume>, <fpage>132</fpage>&#x2013;<lpage>143</lpage>. <pub-id pub-id-type="doi">10.2174/1566523221666210622164133</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jeon</surname>
<given-names>Y. J.</given-names>
</name>
<name>
<surname>Hasan</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>H. W.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Manavalan</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Tacos: A novel approach for accurate prediction of cell-specific long noncoding RNAs subcellular localization</article-title>. <source>Brief. Bioinform</source> <volume>23</volume>, <fpage>bbac243</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac243</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karuna Nidhi</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Ganapathy</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Subbiah</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Suvaiyarasan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Karuppasamy</surname>
<given-names>M. P. J. C. B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>GenNBPSeq: Online web server to generate never born protein sequences using toeplitz matrix approach with structure analysis</article-title>. <source>Curr. Bioinform.</source> <volume>17</volume>, <fpage>565</fpage>&#x2013;<lpage>577</lpage>. <pub-id pub-id-type="doi">10.2174/1574893617666220519110154</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>H. J. J. O. T. B.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>The modified Mahalanobis discriminant for predicting outer membrane proteins by using Chou&#x27;s pseudo amino acid composition</article-title>. <source>J. Theor. Biol.</source> <volume>252</volume>, <fpage>350</fpage>&#x2013;<lpage>356</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2008.02.004</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Manavalan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Basith</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shin</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>G. J. B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>mAHTPred: a sequence-based meta-predictor for improving the prediction of anti-hypertensive peptides using effective feature representation</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>2757</fpage>&#x2013;<lpage>2765</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty1047</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Manavalan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Patra</surname>
<given-names>M. C. J. J. O. M. B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Mlcpp 2.0: An updated cell-penetrating peptides and their uptake efficiency predictor</article-title>. <source>J. Mol. Biol.</source> <volume>434</volume>, <fpage>167604</fpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2022.167604</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>K.-J.</given-names>
</name>
<name>
<surname>Gromiha</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Horton</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Suwa</surname>
<given-names>M. J. B.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Discrimination of outer membrane proteins using support vector machines</article-title>. <source>Bioinformatics</source> <volume>21</volume>, <fpage>4223</fpage>&#x2013;<lpage>4229</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bti697</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>SCovid: Single-cell atlases for exposing molecular characteristics of COVID-19 across 10 human tissues</article-title>. <source>Nucleic acids Res.</source> <volume>50</volume>, <fpage>D867</fpage>&#x2013;<lpage>D874</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab881</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rollauer</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Sooreshjani</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Noinaj</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Buchanan</surname>
<given-names>S. K. J. P. T. O. T. R. S. B. B. S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Outer membrane protein biogenesis in Gram-negative bacteria</article-title>. <source>Philos. Trans. R. Soc. Lond B Biol. Sci.</source> <volume>370</volume>, <fpage>20150023</fpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rout</surname>
<given-names>R. K.</given-names>
</name>
<name>
<surname>Hassan</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Sheikh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Umer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sahoo</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Gandomi</surname>
<given-names>A. H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Feature-extraction and analysis based on spatial distribution of amino acids for SARS-CoV-2 Protein sequences</article-title>. <source>Comput. Biol. Med.</source> <volume>141</volume>, <fpage>105024</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2021.105024</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>H.-B.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>K.-C. J. a. B.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>PseAAC: A flexible web server for generating various kinds of protein pseudo amino acid composition</article-title>. <source>Anal. Biochem.</source> <volume>373</volume>, <fpage>386</fpage>&#x2013;<lpage>388</lpage>. <pub-id pub-id-type="doi">10.1016/j.ab.2007.10.012</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shoombuatong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Basith</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pitti</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Manavalan</surname>
<given-names>B. J. J. O. M. B.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Throne: A new approach for accurate prediction of human RNA N7-methylguanosine sites</article-title>. <source>J. Mol. Biol.</source> <volume>434</volume>, <fpage>167549</fpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2022.167549</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shoombuatong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Basith</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pitti</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Manavalan</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Throne: A new approach for accurate prediction of human rna N7-methylguanosine sites</article-title>. <source>J. Mol. Biol.</source> <volume>434</volume>, <fpage>167549</fpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2022.167549</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.-L.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.-S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.-H.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Ppd: A manually curated database for experimentally verified prokaryotic promoters</article-title>. <source>J. Mol. Biol.</source> <volume>433</volume>, <fpage>166860</fpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2021.166860</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>PSnoD: Identifying potential snoRNA-disease associations based on bounded nuclear norm regularization</article-title>. <source>Brief. Bioinform.</source> <volume>23</volume>, <fpage>bbac240</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac240</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tabaie</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Orenstein</surname>
<given-names>E. W.</given-names>
</name>
<name>
<surname>Nemati</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Basu</surname>
<given-names>R. K.</given-names>
</name>
<name>
<surname>Kandaswamy</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Clifford</surname>
<given-names>G. D.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Predicting presumed serious infection among hospitalized children on central venous lines with machine learning</article-title>. <source>Comput. Biol. Med.</source> <volume>132</volume>, <fpage>104289</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2021.104289</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tran</surname>
<given-names>H. V.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>Q. H. J. C. B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>iAnt: combination of convolutional neural network and random Forest models using PSSM and BERT features to identify antioxidant proteins</article-title>. <source>Curr. Bioinform.</source> <volume>17</volume>, <fpage>184</fpage>&#x2013;<lpage>195</lpage>. <pub-id pub-id-type="doi">10.2174/1574893616666210820095144</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vapnik</surname>
<given-names>V. N. J. A.</given-names>
</name>
<name>
<surname>Control</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Complete statistical theory of learning</article-title>. <source>Inf. Fusion</source> <volume>80</volume>, <fpage>1949</fpage>&#x2013;<lpage>1975</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>microbioTA: an atlas of the microbiome in multiple disease tissues of <italic>Homo sapiens</italic> and <italic>Mus musculus</italic>
</article-title>. <source>Nucleic acids Res.</source> <volume>51</volume>, <fpage>D1345</fpage>&#x2013;<lpage>D1352</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkac851</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L. J. P.</given-names>
</name>
<name>
<surname>Letters</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Discrimination of outer membrane proteins by a new measure of information discrepancy</article-title>. <source>Protein Pept. Lett.</source> <volume>14</volume>, <fpage>37</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.2174/092986607779117254</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ning</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Analysis and modeling of myopia-related factors based on questionnaire survey</article-title>. <source>Comput. Biol. Med.</source> <volume>150</volume>, <fpage>106162</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106162</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y. J. a. A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Discrimination of outer membrane proteins using a K-nearest neighbor method</article-title>. <source>Amino Acids</source> <volume>35</volume>, <fpage>65</fpage>&#x2013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.1007/s00726-007-0628-7</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Risk prediction of diabetes: Big data mining with fusion of multifarious physical examination indicators</article-title>. <source>Inf. Fusion</source> <volume>75</volume>, <fpage>140</fpage>&#x2013;<lpage>149</lpage>. <pub-id pub-id-type="doi">10.1016/j.inffus.2021.02.015</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>DeepIDC: A prediction framework of injectable drug combination based on heterogeneous information and deep learning</article-title>. <source>Clin. Pharmacokinet.</source> <volume>61</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1007/s40262-022-01180-9</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Z.-A.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.-K.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>P.-F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Recent advances in predicting protein-lncRNA interactions using machine learning methods</article-title>. <source>Curr. Gene Ther.</source> <volume>22</volume>, <fpage>228</fpage>&#x2013;<lpage>244</lpage>. <pub-id pub-id-type="doi">10.2174/1566523221666210712190718</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Identification of chronic hypersensitivity pneumonitis biomarkers with machine learning and differential Co-expression analysis</article-title>. <source>Curr. Gene Ther.</source> <volume>21</volume>, <fpage>299</fpage>&#x2013;<lpage>303</lpage>. <pub-id pub-id-type="doi">10.2174/1566523220666201208093325</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ju</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>D. J. C. B.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Distance-based support vector machine to predict DNA N6-methyladenine modification</article-title>. <source>Curr. Bioinform.</source> <volume>17</volume>, <fpage>473</fpage>&#x2013;<lpage>482</lpage>. <pub-id pub-id-type="doi">10.2174/1574893617666220404145517</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>H. J. C. O.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Exosomal non-coding RNAs: New insights into the biology of hepatocellular carcinoma</article-title>. <source>Curr. Oncol.</source> <volume>29</volume>, <fpage>5383</fpage>&#x2013;<lpage>5406</lpage>. <pub-id pub-id-type="doi">10.3390/curroncol29080427</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.-Y.</given-names>
</name>
<name>
<surname>Ning</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Futamura</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sakurai</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2022c</year>). <article-title>iLoc-miRNA: extracellular/intracellular miRNA prediction using deep BiLSTM with attention mechanism</article-title>. <source>Brief. Bioinform.</source> <volume>23</volume>, <fpage>bbac395</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac395</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J. J. C. B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Multivariate information fusion for identifying antifungal peptides with Hilbert-Schmidt Independence Criterion</article-title>. <source>Curr. Bioinform.</source> <volume>17</volume>, <fpage>89</fpage>&#x2013;<lpage>100</lpage>. <pub-id pub-id-type="doi">10.2174/1574893616666210727161003</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Identification of gene signature associated with type 2 diabetes mellitus by integrating mutation and expression data</article-title>. <source>Curr. Gene Ther.</source> <volume>22</volume>, <fpage>51</fpage>&#x2013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.2174/1566523221666210707140839</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>