<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioeng. Biotechnol.</journal-id>
<journal-title>Frontiers in Bioengineering and Biotechnology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioeng. Biotechnol.</abbrev-journal-title>
<issn pub-type="epub">2296-4185</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">752658</article-id>
<article-id pub-id-type="doi">10.3389/fbioe.2021.752658</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioengineering and Biotechnology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>ECM-LSE: Prediction of Extracellular Matrix Proteins Using Deep Latent Space Encoding of k-Spaced Amino Acid Pairs</article-title>
<alt-title alt-title-type="left-running-head">Al-Saggaf et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">ECM-LSE</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Al-Saggaf</surname>
<given-names>Ubaid M.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1453731/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Usman</surname>
<given-names>Muhammad</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/965018/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Naseem</surname>
<given-names>Imran</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1485002/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Moinuddin</surname>
<given-names>Muhammad</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1491490/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jiman</surname>
<given-names>Ahmad A.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1507441/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Alsaggaf</surname>
<given-names>Mohammed U.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Alshoubaki</surname>
<given-names>Hitham K.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1435806/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Khan</surname>
<given-names>Shujaat</given-names>
</name>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1428288/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<label>
<sup>1</sup>
</label>Center of Excellence in Intelligent Engineering Systems, King Abdulaziz University, <addr-line>Jeddah</addr-line>, <country>Saudi Arabia</country>
</aff>
<aff id="aff2">
<label>
<sup>2</sup>
</label>Electrical and Computer Engineering Department, King Abdulaziz University, <addr-line>Jeddah</addr-line>, <country>Saudi Arabia</country>
</aff>
<aff id="aff3">
<label>
<sup>3</sup>
</label>Department of Computer Engineering, Chosun University, <addr-line>Gwangju</addr-line>, <country>South Korea</country>
</aff>
<aff id="aff4">
<label>
<sup>4</sup>
</label>Research and Development, Love For Data, <addr-line>Karachi</addr-line>, <country>Pakistan</country>
</aff>
<aff id="aff5">
<label>
<sup>5</sup>
</label>School of Electrical, Electronic and Computer Engineering, The University of Western Australia, <addr-line>Perth</addr-line>, <addr-line>WA</addr-line>, <country>Australia</country>
</aff>
<aff id="aff6">
<label>
<sup>6</sup>
</label>College of Engineering, Karachi Institute of Economics and Technology, Korangi Creek, <addr-line>Karachi</addr-line>, <country>Pakistan</country>
</aff>
<aff id="aff7">
<label>
<sup>7</sup>
</label>Department of Radiology, Faculty of Medicine, King Abdulaziz University, <addr-line>Jeddah</addr-line>, <country>Saudi Arabia</country>
</aff>
<aff id="aff8">
<label>
<sup>8</sup>
</label>Department of Bio and Brain Engineering, <addr-line>Daejeon</addr-line>, <country>South Korea</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/476229/overview">Zhiguang Zhu</ext-link>, Tianjin Institute of Industrial Biotechnology, (CAS), China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/54051/overview">Mario Andrea Marchisio</ext-link>, Tianjin University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1467956/overview">Ge Qu</ext-link>, Tianjin Institute of Industrial Biotechnology, (CAS), China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Shujaat Khan&#x2009;, <email>shujaat@kaist.ac.kr</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Synthetic Biology, a section of the journal Frontiers in Bioengineering and Biotechnology</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>14</day>
<month>10</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>9</volume>
<elocation-id>752658</elocation-id>
<history>
<date date-type="received">
<day>03</day>
<month>08</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>09</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2021 Al-Saggaf, Usman, Naseem, Moinuddin, Jiman, Alsaggaf, Alshoubaki and Khan.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Al-Saggaf, Usman, Naseem, Moinuddin, Jiman, Alsaggaf, Alshoubaki and Khan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>Extracelluar matrix (ECM) proteins create complex networks of macromolecules which fill-in the extracellular spaces of living tissues. They provide structural support and play an important role in maintaining cellular functions. Identification of ECM proteins can play a vital role in studying various types of diseases. Conventional wet lab&#x2013;based methods are reliable; however, they are expensive and time consuming and are, therefore, not scalable. In this research, we propose a sequence-based novel machine learning approach for the prediction of ECM proteins. In the proposed method, composition of k-spaced amino acid pair (CKSAAP) features are encoded into a classifiable latent space (LS) with the help of deep latent space encoding (LSE). A comprehensive ablation analysis is conducted for performance evaluation of the proposed method. Results are compared with other state-of-the-art methods on the benchmark dataset, and the proposed ECM-LSE approach has shown to comprehensively outperform the contemporary methods.</p>
</abstract>
<kwd-group>
<kwd>extracellular matrix (ECM)</kwd>
<kwd>auto-encoder</kwd>
<kwd>composition of k-spaced amino acid pair (CKSAAP)</kwd>
<kwd>latent space learning</kwd>
<kwd>neural network</kwd>
<kwd>classification</kwd>
<kwd>amino acid composition (AAC)</kwd>
</kwd-group>
<contract-num rid="cn001">IFPHI-139-135-2020</contract-num>
<contract-sponsor id="cn001">King Abdulaziz University<named-content content-type="fundref-id">10.13039/501100004054</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Extracelluar matrix (ECM) is a network of fibrous proteins filled in the extracellular spaces of living tissues to provide structural support for the cells (<xref ref-type="bibr" rid="B29">Karag&#xf6;z et&#x20;al., 2021)</xref>. It is significant for cell functionality and plays an important role in the physiological dynamics. ECMs are also responsible for the promotion of vital cellular processes, including differentiation, adhesion, proliferation, apoptosis, and migration (<xref ref-type="bibr" rid="B33">Klavert and van der Eerden, 2021</xref>; <xref ref-type="bibr" rid="B15">Hiraki et&#x20;al., 2021</xref>; <xref ref-type="bibr" rid="B41">Mathews et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B11">Endo et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B31">Kim et&#x20;al., 2011</xref>). The chemical composition of ECM mainly consists of minerals, proteoglycans, proteins, and water. The proteins in ECM act more like a fibrous material which gives strength to the cells. Several studies have demonstrated that the mutation in the ECM genes can cause severe adverse effects in the cell structure resulting in a number of diseases, including arthritis and cancer (<xref ref-type="bibr" rid="B32">Kizawa et&#x20;al., 2005</xref>; <xref ref-type="bibr" rid="B19">Hu et&#x20;al., 2007</xref>).</p>
<p>Functional research on ECM protein has resulted in the development of useful biomaterials which are used in many fields of medicine, such as tissue engineering and cell therapy (<xref ref-type="bibr" rid="B40">Ma et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B13">Gonzalez-Pujana et&#x20;al., 2019)</xref>. Proteins, in general, are active elements and play a variety of roles depending on their residing location in a cell. Likewise, the functionality of the ECM varies with the change in the proteins. The problem of protein localization is therefore considered to be an important step toward the understanding of protein functionality (<xref ref-type="bibr" rid="B16">Horton et&#x20;al., 2007)</xref>. Identification of subcellular location is however considered to be a nontrivial task and requires extensive experimentation which is prohibitively expensive. Therefore, a variety of computational methods have been developed to facilitate the process (<xref ref-type="bibr" rid="B51">Ras-Carmona et&#x20;al., 2021</xref>; <xref ref-type="bibr" rid="B58">Wang et&#x20;al., 2021</xref>; <xref ref-type="bibr" rid="B5">Chou, 2011)</xref>. In particular, for different species of plants, animals, and microorganisms, a number of useful techniques have been explored (<xref ref-type="bibr" rid="B70">Zhao et&#x20;al., 2021</xref>; <xref ref-type="bibr" rid="B18">Hou et&#x20;al., 2021</xref>; <xref ref-type="bibr" rid="B6">Chou et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B47">Otzen et&#x20;al., 2021</xref>; <xref ref-type="bibr" rid="B61">Wu et&#x20;al., 2011</xref>; <xref ref-type="bibr" rid="B1">Asim et&#x20;al., 2021</xref>; <xref ref-type="bibr" rid="B63">Xiao et&#x20;al., 2011</xref>; <xref ref-type="bibr" rid="B53">Shen et&#x20;al., 2021</xref>; <xref ref-type="bibr" rid="B60">Wu et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B35">Lewis et&#x20;al., 2014)</xref>. Bioinformatics methods, with the aid of machine learning algorithms, have demonstrated adequate performance for a variety of applications. A detailed review of computational methods to classify secreted proteins has been provided by <xref ref-type="bibr" rid="B34">Klee and Sosa (2007)</xref>. Typically, three aspects are focused on the development of a computational method: 1) feature extraction&#x2014;in which the peptide sequence is translated/encoded into a numerical format to make them readable by the model, 2) feature selection&#x2014;which is concerned with the removal of the redundant information from the feature space and results in the model&#x2019;s robustness, and 3) model construction and evaluation&#x2014;which includes development of a prediction model, followed by training and testing steps to evaluate performance.</p>
<p>The first benchmark <italic>in-silico</italic> approach to predict the extracellular proteins was presented by <xref ref-type="bibr" rid="B24">Jung et&#x20;al. (2010)</xref>and was named as ECM protein prediction (ECMPP). The research used the feature augmentation method and crafted a feature set of 91 attributes. One of the limitations of the study was the use of a small dataset for performance evaluation; also, only the receiver-operating characteristics (ROC) were used for the performance evaluation. Since then, many researchers have paid attention toward the development of machine learning methods for ECM prediction. As extracellular matrix proteins are linked to the outer surface of the cell, they have close association with its secretory mechanism and are naturally associated with the secretary proteins. Therefore, it is reasonable to consider possible ECM candidates as a subset of secretary proteins (<xref ref-type="bibr" rid="B27">Kandaswamy et&#x20;al., 2010</xref>; <xref ref-type="bibr" rid="B10">Bendtsen et&#x20;al., 2004</xref>; <xref ref-type="bibr" rid="B17">Horton et&#x20;al., 2006</xref>). Based on this knowledge, <xref ref-type="bibr" rid="B28">Kandaswamy et&#x20;al. (2013</xref>) improved the ECM prediction method and presented EcmPred. EcmPred (<xref ref-type="bibr" rid="B28">Kandaswamy et&#x20;al., 2013</xref>) used a random forest (RF)&#x2013;based classifier which was trained on the combination of sequence-derived properties of the proteins including individual and group frequencies of amino acids with the physicochemical properties. Another method named prediction of ECM (PECM) (<xref ref-type="bibr" rid="B69">Zhang et&#x20;al., 2014</xref>) utilized a handcrafted feature set designed by the combination of the most discriminative attributes of the protein sequences including evolutionary and structural information as well as the physicochemical properties of the peptide sequences. An incremental feature selection (IFS) method was employed for the selection of optimal features which were used to train a support vector machine (SVM)&#x2013;based classifier. Several other methods have also been proposed to serve the task of ECM prediction. None of them, however, focuses on the encoding of sequence-driven feature into a classifiable latent-space (LS). The primary objective of latent space&#x2013;based learning is to design a reduced feature space for clustering of proteins. The LS is, therefore, a representation of the input signal in a reduced space. The latent-space encoding (LSE) is based on an assumption of a low-rank input (i.e. highly redundant) which can be compressed to a low dimensional signal using LSE. The process is considered to be reversible as the original signal could be reconstructed from the LS. The details of LS and LSE have been provided in the <xref ref-type="sec" rid="s2-4">Subsection&#x20;2.4</xref>.</p>
<p>Development of a feature space and selection of the best features are fundamental steps in designing machine learning models (<xref ref-type="bibr" rid="B39">Lyu et&#x20;al., 2021</xref>). In particular, for the protein sequence classification task, a variety of feature extraction techniques have been proposed including amino acid composition (AAC), dipeptide composition (DPC), N-segmented sequence features, physicochemical composition, and secondary structure features (<xref ref-type="bibr" rid="B45">Naseem et&#x20;al., 2017</xref>; <xref ref-type="bibr" rid="B30">Khan et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B26">Kandaswamy et&#x20;al., 2011</xref>). The sole purpose of each feature extraction technique is to encode maximum useful information from a variable length protein sequence into a fixed-sized vector. In the recent past, inspired by the success of deep long short-term memory (LSTM) models, some approaches similar to word2vec (<xref ref-type="bibr" rid="B43">Mikolov et&#x20;al., 2013</xref>) have been proposed to successfully learn latent space encoding directly from variable length sequences (<xref ref-type="bibr" rid="B9">Ding et&#x20;al., 2019</xref>). The direct sequence to latent space encoding method produces good generalization models (<xref ref-type="bibr" rid="B67">Zemouri, 2020</xref>); however, they usually rely on the availability of a large training dataset. Furthermore, the direct extraction of latent space features from a limited number of sequences such as, bioluminescence (<xref ref-type="bibr" rid="B68">Zhang et&#x20;al., 2021</xref>), antioxidant (<xref ref-type="bibr" rid="B46">Olsen et&#x20;al., 2020</xref>), ECM (<xref ref-type="bibr" rid="B25">Kabir et&#x20;al., 2018</xref>), antifreeze proteins (AFPs) (<xref ref-type="bibr" rid="B26">Kandaswamy et&#x20;al., 2011</xref>), or other classes of proteins is a challenging problem. In this study, we propose a hybrid approach where all proteins are first encoded into a large feature set obtained through composition of <italic>k</italic>-spaced amino acid pair encoding. A latent space representation of composition of <italic>k</italic>-spaced amino acid pairs (CKSAAP) is learned which can help to design a robust classifier. This eliminates the need for separately developing the classifier and the feature extraction modules, and a stand-alone model effectively learns the distinguishing characteristics of classes on a lower dimensional feature&#x20;space.</p>
<p>The rest of the article is organized as follows: the classification framework of the proposed method is presented in <xref ref-type="sec" rid="s2">Section 2</xref>, followed by the extensive experimentation and discussion in <xref ref-type="sec" rid="s3">Section 3</xref>, and the study is concluded in <xref ref-type="sec" rid="s4">Section&#x20;4</xref>.</p>
</sec>
<sec id="s2">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 Evaluation Metrics</title>
<p>For proper evaluation of the proposed model, a number of standard performance metrics have been used. The most intuitive performance measure is accuracy; however, for a highly imbalanced dataset (which is the case here), accuracy is not reflective of true performance. Therefore, various evaluation parameters, such as sensitivity, specificity, and Matthew&#x2019;s correlation coefficient (MCC) are reported. Youden&#x2019;s index and balanced accuracy are also considered to be important evaluation metrics for imbalanced data and are, therefore, extensively explored in this research.</p>
</sec>
<sec id="s2-2">
<title>2.2 Dataset</title>
<p>To design the proposed method, we used the benchmark dataset provided in <xref ref-type="bibr" rid="B28">Kandaswamy et&#x20;al. (2013)</xref>. The dataset consists of 445 ECM proteins and 3,327&#x20;non-ECM proteins. The 445 ECM proteins were curated from Swiss-Prot release 67 by first filtering 1103 ECM proteins from the pool of 17,233&#x20;metazoan-secreted protein sequences. Similarly, the negative dataset of 16,130 proteins were curated from secretory proteins that are annotated as non-ECM. Later, 445 ECM and 4,187&#x20;non-ECM nonhomologous sequences were further filtered out with the help of a clustering method (<xref ref-type="bibr" rid="B37">Li et&#x20;al., 2001</xref>) by removing the sequences which showed 70<italic>%</italic> or higher similarity.</p>
</sec>
<sec id="s2-3">
<title>2.3 Feature Extraction</title>
<sec id="s2-3-1">
<title>2.3.1 Composition of K-Spaced Amino Acid Pairs</title>
<p>One of the fundamental steps in designing a machine learning approach is the transformation of protein sequences to a numerical format. Several methods of this transformation exist and the resultant encoded vectors of the sequences are treated as the features. The common approach practiced by several researchers is to acquire various features of the same sequence by employing different encoding schemes, and their combination is utilized for training the machine learning algorithm. This laborious approach has resulted in the performance enhancement of some classifiers (<xref ref-type="bibr" rid="B66">Yu and Lu, 2011</xref>; <xref ref-type="bibr" rid="B64">Xiaowei et&#x20;al., 2012</xref>; <xref ref-type="bibr" rid="B65">Yang et&#x20;al., 2015</xref>; <xref ref-type="bibr" rid="B62">Xiao et&#x20;al., 2016</xref>); however, some recent studies show that utilizing a single expedient-encoding scheme such as CKSAAP, which captures both short- and long-range interaction information between residues along the sequence, can result in an equally improved classification performance (<xref ref-type="bibr" rid="B22">Ju and Wang, 2018</xref>; <xref ref-type="bibr" rid="B2">Chen et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B56">Usman and Lee, 2019</xref>).</p>
<p>The CKSAAP scheme works on the simple principle of counting the occurrence frequencies of <italic>k</italic>-spaced amino acid pairs in the protein sequence. Each <italic>k</italic>-spaced amino acid pair represents the residue pair separated by any arbitrary number <italic>k</italic> (<italic>j</italic>&#x20;&#x3d; 0, 1, 2&#x20;&#x2026; <italic>k</italic>) of amino acid residues. For <italic>k</italic>&#x20;&#x3d; 0, the encoding is similar to the DPC, in which protein sequence of 20 types of amino acids yields a feature vector of (20 &#xd7; 20) &#x3d; 400 types of amino acid pairs (<italic>i</italic>.<italic>e</italic>., <italic>AA</italic>, <italic>AC</italic>, <italic>AD</italic>, &#x2026; <italic>YY</italic>)<sub>400</sub>. In earlier studies it has been suggested that the DPC and higher-order peptide features can be used to design a robust protein sequence classifier (<xref ref-type="bibr" rid="B26">Kandaswamy et&#x20;al., 2011</xref>; <xref ref-type="bibr" rid="B30">Khan et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B50">Pratiwi et&#x20;al., 2017</xref>). From <xref ref-type="fig" rid="F1">Figure&#x20;1</xref>, it can be seen that for higher values of <italic>k</italic>, substantial neighborhood information is gathered for large peptide pairs. For instance <italic>k</italic>&#x20;&#x3d; 2, three feature segments, each having a length of 400, are obtained. These are then concatenated to get the final feature vector of length (<italic>k</italic>&#x20;&#x2b; 1) &#xd7; 400. The graphical representation of the CKSAAP feature vector obtained with <italic>k</italic>&#x20;&#x3d; 2 has been depicted in <xref ref-type="fig" rid="F1">Figure&#x20;1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>CKSAAP feature extraction mechanism for <italic>k</italic>&#x20;&#x3d; 2. Extracted from <xref ref-type="bibr" rid="B55">Usman et&#x20;al. (2020)</xref>.</p>
</caption>
<graphic xlink:href="fbioe-09-752658-g001.tif"/>
</fig>
<p>This efficient method of encoding has, therefore, been favored by a number of researchers in various applications of computational biology including the prediction of anticancer peptides (<xref ref-type="bibr" rid="B36">Li et&#x20;al., 2020</xref>), DNA, and several other binding sites (<xref ref-type="bibr" rid="B23">Ju and Wang, 2020</xref>; <xref ref-type="bibr" rid="B38">Lyu et&#x20;al., 2020</xref>). Many adaptations of CKSAAP encoding scheme have utilized only the features generated by a single <italic>k</italic> value. In this research, we aim to find the optimal value of <italic>k</italic> by analyzing different combinations of the features generated by CKSAAP, and details are presented in <xref ref-type="sec" rid="s3-1">Subsection&#x20;3.1</xref>.</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Latent Space Learning for ECM Classification</title>
<p>Feature representation ability of the CKSAAP improves with large values of the parameter <italic>k</italic>, which is expected to result in a more robust model (<xref ref-type="bibr" rid="B49">Park et&#x20;al., 2020b</xref>; <xref ref-type="bibr" rid="B56">Usman and Lee, 2019</xref>; <xref ref-type="bibr" rid="B59">Wu et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B3">Chen et&#x20;al., 2017</xref>). However, the model utilizing a large number of features is susceptible to noise, resulting in a degraded performance. Furthermore, training the model on a large number of features not only results in an increased training time and complexity but is also prone to overfitting. To which end, feature selection/engineering, which involves the selection of most significant features, has to be employed. Feature selection techniques are broadly categorized into two types: 1) supervised methods, which remove the irrelevant features based on a target variable, and 2) unsupervised methods, which use correlation techniques to remove redundant information. A number of methods for feature selection have been proposed in the literature, including minimum redundancy maximum relevance (mRMR) (<xref ref-type="bibr" rid="B14">Peng et&#x20;al., 2005</xref>), student&#x2019;s t&#x20;test (<xref ref-type="bibr" rid="B54">Student, 1908</xref>), info-gain (<xref ref-type="bibr" rid="B44">Mitchell et&#x20;al., 1997</xref>), and generalized variant of strictly standardized mean difference (GSSMD) (<xref ref-type="bibr" rid="B48">Park et&#x20;al., 2020a</xref>). Another useful method is to map the original data into a lower-order dimensional space through some transformation function. The eigen-space transformation or the principal component analysis method (PCA) (<xref ref-type="bibr" rid="B21">Jolliffe, 1986</xref>) is considered to be the benchmark method in this context. Other approaches such as an independent component analysis (ICA) (<xref ref-type="bibr" rid="B7">Comon, 1994</xref>), a kernel principal component analysis (KPCA) (<xref ref-type="bibr" rid="B52">Sch&#xf6;lkopf et&#x20;al., 1998</xref>), uniform manifold approximation and projection (UMAP) (<xref ref-type="bibr" rid="B42">McInnes et&#x20;al., 2018</xref>), and t-distributed stochastic neighbor embedding (t-SNE) (<xref ref-type="bibr" rid="B57">Van der Maaten and Hinton, 2008</xref>) are also being successfully used to deal with the curse of dimensionality.</p>
<p>Most of the techniques mentioned above are unsupervised in nature. To address this issue, we propose to use a novel approach called a deep latent space encoding (DeepLSE) classifier for the latent space encoding based on an auto-encoder. Latent space refers to the representation of compressed data in which similar points would be in a close group, as shown in <xref ref-type="fig" rid="F5">Figure&#x20;5</xref>. Similar samples tend to have common significance, which can be packaged into the latent space representation of the raw data. Thus, as the dimensions are reduced, the redundant information from the input samples is removed, leaving only the most important features of the data. In other words, the method can learn a compact representation of feature space and remove the noisy or potentially confusing information which is good for both the classification and reconstruction tasks. This ensures that the encoded features truly represent the sample information. The DeepLSE method has been found to be an impressive method for the feature space reduction and has outperformed other approaches in relatively similar tasks such as AFP-LSE (<xref ref-type="bibr" rid="B55">Usman et&#x20;al., 2020</xref>) and E3-targetPred (<xref ref-type="bibr" rid="B49">Park et&#x20;al., 2020b</xref>). The architecture of the proposed method is depicted in <xref ref-type="fig" rid="F2">Figure&#x20;2</xref> named as ECM-LSE.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Proposed DeepLSE architecture for ECM classification. The model comprises of an encoder, a decoder, and a classifier module. The encoder consists of an input layer and two hidden layers that embed input features to latent variables (<italic>LVs</italic>). The decoder architecture is mirror symmetry of the encoder which uses <italic>LVs</italic> as its input and generates the decoded output. The classification module uses latent space features as its input and four layers of fully connected neurons. Each hidden layer has 10 neurons, except for the last layer which produces a one-hot&#x2013;encoded output of ECM/non-ECM&#x20;class.</p>
</caption>
<graphic xlink:href="fbioe-09-752658-g002.tif"/>
</fig>
<sec id="s2-4-1">
<title>2.4.1 Network Specifications</title>
<p>The architecture of the proposed ECM-LSE network is composed of two modules: 1) an auto-encoder module and 2) a classification module.</p>
<sec id="s2-4-1-1">
<title>2.4.1.1&#x20;Auto-Encoder Module</title>
<p>The auto-encoder is a type of neural network that can act as an identity function. It is used to find the representation of the input signal in a reduced dimensional space, known as the latent space. The principle of latent space&#x2013;based representation is an assumption that the input signal has a low-rank. The auto-encoder network has a decoder that tries to regenerate the input from the latent space variables. During the training of an auto-encoder, the model is forced to become an identity function. Due to which only the relevant features of the data are learned in a compressed representation. This compressed representation has sufficient information for accurate reconstruction of the original input signal. The number of hidden layers and the number of neurons in each layer of the encoder and decoder are varied to obtain reasonable performance. In this research, the encoder and decoder are composed of three layers each, including two hidden layers. The number of neurons in the input layer of the encoder is equal to the length of the attribute vector. The number of neurons in the first and second hidden layers is set to be 50 and 10, respectively. The decoder is a mirror symmetry of the encoder. The number of neurons in the output layer of the decoder is equal to the length of the attribute vector. The number of neurons in the latent space is systematically altered to obtain the best performance for which we designed an ablation study discussed in <xref ref-type="sec" rid="s3-1">Section 3.1</xref>. All hidden layers of the auto-encoder module are equipped with batch normalization, 30<italic>%</italic> dropout, and a rectified linear unit (ReLU) activation function. The latent space layer uses sigmoid activation function without any batch normalization and dropout.</p>
</sec>
<sec id="s2-4-1-2">
<title>2.4.1.2 Classification Module</title>
<p>The output of the encoder module (latent variables) is used as an input to the classification module. The classifier module shown in <xref ref-type="fig" rid="F2">Figure&#x20;2</xref> consists of four layers (three hidden and one output layer). All hidden layers consist of 10 neurons and a ReLU activation function. The last layer consists of two neurons representing the positive (ECM) and the negative (non-ECM) classes. For decision making, softmax activation function was used at the output&#x20;layer.</p>
</sec>
</sec>
</sec>
</sec>
<sec id="s3">
<title>3 Results</title>
<p>To develop a neural network model, the benchmark dataset was divided into the train, validation, and test datasets. For training, we formed a dataset consisting of 540 samples with equal number of ECMs and non-ECM protein samples. These were randomly selected from the pool of 445 ECMs and 3,327&#x20;non-ECMs, since the available dataset is very small, and it is highly likely that the model would suffer from the overfitting problem. To avoid such situation, we employed regularization techniques such as early stopping, dropout, batch normalization, and DeepLSE-based feature encoding. Furthermore, the validation dataset was also used with the aim of designing a generalized classifier module. The validation dataset consists of 30 ECMs and 810&#x20;non-ECMs randomly selected from the remaining 175 ECMs and 3,182&#x20;non-ECMs, respectively. The remaining 145 and 2,247 samples of ECMs and non-ECMs were used in the test dataset. Several model configurations on the basis of the latent space size (<italic>LVs</italic>) and the CKSAAP gap value <italic>k</italic> were evaluated. For each choice of model configuration, the process of model training was repeated 20&#x20;times and mean and standard deviations of performance statistics were reported. In each trial, the weights and bias of the model were randomly initialized. Also, each trial utilized randomly configured subsets from the training, validation, and test dataset. The validation process assisted toward the filtration of the overfitted models, that is, only the models with 75<italic>%</italic> or higher validation balanced accuracy was selected.</p>
<sec id="s3-1">
<title>3.1 Ablation Study</title>
<p>The workflow of the proposed study is aimed to obtain the best classification model based on two variables, that is, the gap between the two amino acid pairs and the number of units in the latent space <italic>LVs</italic>. An ablation study has been designed to acquire models with varying number of aforementioned variables and is depicted in <xref ref-type="fig" rid="F3">Figure&#x20;3</xref> (a). The samples are distributed into training, validation, and test datasets as discussed in the <xref ref-type="sec" rid="s2-2">Subsection 2.2</xref> and are encoded with incrementing values of <italic>k</italic> from 0 to 10. The resultant features are used to train the model with incrementing values of the latent space variables ranging from 2 to 9. As discussed earlier, for each configuration, 20 independent trials are performed and the mean results are computed. A consistent procedure is repeated for all 1,760 trials of the 88 unique model configurations. The model with the best average results is finally selected as the base model to perform prediction and is named as ECM-LSE. In <xref ref-type="table" rid="T1">Table. 1</xref>, the average results of the balanced accuracy have been reported. It can be observed that the model with values of gap <italic>k</italic>&#x20;&#x3d; 8 and latent variables <italic>LV</italic> &#x3d; 7, accounts for the best. The results for the rest of the evaluation parameters are illustrated in the form of surface graphs in <xref ref-type="fig" rid="F4">Figure&#x20;4</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Workflow of the proposed ECM-LSE method.</p>
</caption>
<graphic xlink:href="fbioe-09-752658-g003.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Balanced accuracy results of ablation study on Gap (<italic>k</italic>) and <italic>LV</italic> parameters.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Gap/<italic>LV</italic>
</th>
<th align="center">2</th>
<th align="center">3</th>
<th align="center">4</th>
<th align="center">5</th>
<th align="center">6</th>
<th align="center">7</th>
<th align="center">8</th>
<th align="center">9</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 0</td>
<td align="char" char="plusmn">0.779&#x20;&#xb1; 0.022</td>
<td align="char" char="plusmn">0.776&#x20;&#xb1; 0.027</td>
<td align="char" char="plusmn">0.768&#x20;&#xb1; 0.026</td>
<td align="char" char="plusmn">0.758&#x20;&#xb1; 0.034</td>
<td align="char" char="plusmn">0.760&#x20;&#xb1; 0.028</td>
<td align="char" char="plusmn">0.767&#x20;&#xb1; 0.015</td>
<td align="char" char="plusmn">0.769&#x20;&#xb1; 0.029</td>
<td align="char" char="plusmn">0.775&#x20;&#xb1; 0.027</td>
</tr>
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 1</td>
<td align="char" char="plusmn">0.795&#x20;&#xb1; 0.025</td>
<td align="char" char="plusmn">0.786&#x20;&#xb1; 0.030</td>
<td align="char" char="plusmn">0.780&#x20;&#xb1; 0.027</td>
<td align="char" char="plusmn">0.788&#x20;&#xb1; 0.020</td>
<td align="char" char="plusmn">0.785&#x20;&#xb1; 0.030</td>
<td align="char" char="plusmn">0.784&#x20;&#xb1; 0.038</td>
<td align="char" char="plusmn">0.765&#x20;&#xb1; 0.034</td>
<td align="char" char="plusmn">0.783&#x20;&#xb1; 0.030</td>
</tr>
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 2</td>
<td align="char" char="plusmn">0.803&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.788&#x20;&#xb1; 0.036</td>
<td align="char" char="plusmn">0.793&#x20;&#xb1; 0.025</td>
<td align="char" char="plusmn">0.789&#x20;&#xb1; 0.029</td>
<td align="char" char="plusmn">0.793&#x20;&#xb1; 0.024</td>
<td align="char" char="plusmn">0.796&#x20;&#xb1; 0.030</td>
<td align="char" char="plusmn">0.795&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.798&#x20;&#xb1; 0.027</td>
</tr>
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 3</td>
<td align="char" char="plusmn">0.791&#x20;&#xb1; 0.031</td>
<td align="char" char="plusmn">0.797&#x20;&#xb1; 0.029</td>
<td align="char" char="plusmn">0.808&#x20;&#xb1; 0.015</td>
<td align="char" char="plusmn">0.812&#x20;&#xb1; 0.018</td>
<td align="char" char="plusmn">0.814&#x20;&#xb1; 0.028</td>
<td align="char" char="plusmn">0.803&#x20;&#xb1; 0.027</td>
<td align="char" char="plusmn">0.803&#x20;&#xb1; 0.030</td>
<td align="char" char="plusmn">0.799&#x20;&#xb1; 0.32</td>
</tr>
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 4</td>
<td align="char" char="plusmn">0.785&#x20;&#xb1; 0.028</td>
<td align="char" char="plusmn">0.790&#x20;&#xb1; 0.047</td>
<td align="char" char="plusmn">0.809&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.816&#x20;&#xb1; 0.026</td>
<td align="char" char="plusmn">0.797&#x20;&#xb1; 0.029</td>
<td align="char" char="plusmn">0.786&#x20;&#xb1; 0.026</td>
<td align="char" char="plusmn">0.803&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.797&#x20;&#xb1; 0.037</td>
</tr>
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 5</td>
<td align="char" char="plusmn">0.822&#x20;&#xb1; 0.018</td>
<td align="char" char="plusmn">0.799&#x20;&#xb1; 0.032</td>
<td align="char" char="plusmn">0.803&#x20;&#xb1; 0.035</td>
<td align="char" char="plusmn">0.813&#x20;&#xb1; 0.025</td>
<td align="char" char="plusmn">0.800&#x20;&#xb1; 0.031</td>
<td align="char" char="plusmn">0.826&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.802&#x20;&#xb1; 0.023</td>
<td align="char" char="plusmn">0.811&#x20;&#xb1; 0.019</td>
</tr>
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 6</td>
<td align="char" char="plusmn">0.808&#x20;&#xb1; 0.046</td>
<td align="char" char="plusmn">0.805&#x20;&#xb1; 0.023</td>
<td align="char" char="plusmn">0.817&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.814&#x20;&#xb1; 0.026</td>
<td align="char" char="plusmn">0.810&#x20;&#xb1; 0.027</td>
<td align="char" char="plusmn">0.814&#x20;&#xb1; 0.022</td>
<td align="char" char="plusmn">0.803&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.805&#x20;&#xb1; 0.031</td>
</tr>
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 7</td>
<td align="char" char="plusmn">0.813&#x20;&#xb1; 0.032</td>
<td align="char" char="plusmn">0.824&#x20;&#xb1; 0.033</td>
<td align="char" char="plusmn">0.812&#x20;&#xb1; 0.029</td>
<td align="char" char="plusmn">0.806&#x20;&#xb1; 0.024</td>
<td align="char" char="plusmn">0.824&#x20;&#xb1; 0.027</td>
<td align="char" char="plusmn">0.818&#x20;&#xb1; 0.029</td>
<td align="char" char="plusmn">0.808&#x20;&#xb1; 0.041</td>
<td align="char" char="plusmn">0.801&#x20;&#xb1; 0.022</td>
</tr>
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 8</td>
<td align="char" char="plusmn">0.811&#x20;&#xb1; 0.029</td>
<td align="char" char="plusmn">0.805&#x20;&#xb1; 0.039</td>
<td align="char" char="plusmn">0.807&#x20;&#xb1; 0.034</td>
<td align="char" char="plusmn">0.815&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.816&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.830&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.814&#x20;&#xb1; 0.029</td>
<td align="char" char="plusmn">0.816&#x20;&#xb1; 0.026</td>
</tr>
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 9</td>
<td align="char" char="plusmn">0.796&#x20;&#xb1; 0.034</td>
<td align="char" char="plusmn">0.813&#x20;&#xb1; 0.022</td>
<td align="char" char="plusmn">0.804&#x20;&#xb1; 0.029</td>
<td align="char" char="plusmn">0.814&#x20;&#xb1; 0.026</td>
<td align="char" char="plusmn">0.811&#x20;&#xb1; 0.034</td>
<td align="char" char="plusmn">0.824&#x20;&#xb1; 0.032</td>
<td align="char" char="plusmn">0.809&#x20;&#xb1; 0.025</td>
<td align="char" char="plusmn">0.798&#x20;&#xb1; 0.034</td>
</tr>
<tr>
<td align="left">
<italic>k</italic>&#x20;&#x3d; 10</td>
<td align="char" char="plusmn">0.819&#x20;&#xb1; 0.037</td>
<td align="char" char="plusmn">0.821&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.817&#x20;&#xb1; 0.034</td>
<td align="char" char="plusmn">0.823&#x20;&#xb1; 0.021</td>
<td align="char" char="plusmn">0.817&#x20;&#xb1; 0.027</td>
<td align="char" char="plusmn">0.807&#x20;&#xb1; 0.025</td>
<td align="char" char="plusmn">0.819&#x20;&#xb1; 0.025</td>
<td align="char" char="plusmn">0.816&#x20;&#xb1; 0.031</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Performance statistics surfaces for: <bold>(A)</bold> accuracy, <bold>(B)</bold> MCC, <bold>(C)</bold> balanced-accuracy, <bold>(D)</bold> Youden&#x2019;s Index, <bold>(E)</bold> F1-Score, and <bold>(F)</bold> mean squared error (MSE) in dB.</p>
</caption>
<graphic xlink:href="fbioe-09-752658-g004.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Comparison With Contemporary Approaches</title>
<p>The performance of the proposed model is compared to the benchmark approaches and the findings are reported in <xref ref-type="table" rid="T2">Table 2</xref>. For a fair comparison, only the best reported results of the respective approaches are presented. The performance of the proposed ECM-LSE is compared with the contemporary methods including EcmPred (<xref ref-type="bibr" rid="B28">Kandaswamy et&#x20;al., 2013</xref>), a sparse learning approach for the prediction of ECM (ECMSRC) (<xref ref-type="bibr" rid="B45">Naseem et&#x20;al., 2017</xref>), and PECM (<xref ref-type="bibr" rid="B69">Zhang et&#x20;al., 2014</xref>). In particular, the reported sensitivity, specificity, MCC, Youden&#x2019;s index, and accuracy on the benchmark dataset of EcmPred (<xref ref-type="bibr" rid="B28">Kandaswamy et&#x20;al., 2013</xref>) are compared.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Comparison of the proposed ECM-LSE algorithm with the benchmark machine learning approaches on the test dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="center">Sensitivity (%)</th>
<th align="center">Specificity (%)</th>
<th align="center">MCC</th>
<th align="center">Youden&#x2019;s index</th>
<th align="center">Accuracy (%)</th>
<th align="center">Balanced accuracy (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">EcmPred (<xref ref-type="bibr" rid="B28">Kandaswamy et&#x20;al., 2013)</xref>
</td>
<td align="char" char=".">65.00</td>
<td align="char" char=".">77.00</td>
<td align="char" char=".">0.1910</td>
<td align="char" char=".">0.42</td>
<td align="char" char=".">77.00</td>
<td align="char" char=".">71.00</td>
</tr>
<tr>
<td align="left">ECMSRC (<xref ref-type="bibr" rid="B45">Naseem et&#x20;al., 2017)</xref>
</td>
<td align="char" char=".">74.48</td>
<td align="char" char=".">81.31</td>
<td align="char" char=".">0.2560</td>
<td align="char" char=".">0.56</td>
<td align="char" char=".">81.06</td>
<td align="char" char=".">77.90</td>
</tr>
<tr>
<td align="left">PECM (<xref ref-type="bibr" rid="B69">Zhang et&#x20;al., 2014)</xref>
</td>
<td align="char" char=".">75.86</td>
<td align="char" char=".">
<bold>86.88</bold>
</td>
<td align="char" char=".">0.3143</td>
<td align="char" char=".">0.63</td>
<td align="char" char=".">
<bold>86.52</bold>
</td>
<td align="char" char=".">81.37</td>
</tr>
<tr>
<td align="left">
<bold>ECM-LSE</bold>
</td>
<td align="char" char=".">
<bold>84.14</bold>
</td>
<td align="char" char=".">86.45</td>
<td align="char" char=".">
<bold>0.3906</bold>
</td>
<td align="char" char=".">
<bold>0.71</bold>
</td>
<td align="char" char=".">86.35</td>
<td align="char" char=".">
<bold>85.30</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold-face represent best performance.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The results clearly show that the proposed method has better balanced accuracy as compared to the contemporary approaches. In particular, the proposed ECM-LSE method achieves the highest sensitivity of 84.14<italic>%</italic> outperforming the best competitor (PECM) by a margin of 10.91<italic>%</italic>. The specificity value achieved by the proposed ECM-LSE also compares favorably with other methods, which confirms the balanced unbiased learning effect. It is noteworthy to point out that the accuracy metric cannot provide true fitness of the models given the skewed distribution of test dataset toward the negative (non-ECM) class. Any model with all negative predictions can achieve <inline-formula id="inf1">
<mml:math id="m1">
<mml:mn>100</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2247</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>145</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2247</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>93.94</mml:mn>
<mml:mi>%</mml:mi>
</mml:math>
</inline-formula> accuracy easily. As discussed in <xref ref-type="sec" rid="s2-1">Subsection 2.1</xref>, the parameters of balanced accuracy, MCC, and Youden&#x2019;s index are considered more reliable in the case of imbalanced dataset. Therefore, despite achieving 86.35<italic>%</italic> test accuracy, which is 0.17<italic>%</italic> lower than the PECM, better balanced accuracy and Youden&#x2019;s index values, which is 3.93<italic>%</italic> and 0.08 units higher, respectively, demonstrate the superiority of the proposed method. Similarly, the MCC value achieved by ECM-LSE is 7.63<italic>%</italic> higher than the PECM method. MCC metric is preferred for accuracy and is considered as more reliable statistical parameter because it produces a higher value only if the classifier achieved good results in all four categories of the confusion matrix (<xref ref-type="bibr" rid="B4">Chicco and Jurman, 2020</xref>). In general, the proposed ECM-LSE approach has shown to comprehensively outperform the contemporary methods in all aspects of balanced and unbiased prediction performance.</p>
<p>Furthermore, unlike contemporary methods where handcrafted embedding schemes are utilized for separately developing the classifier and the feature extraction modules, the proposed ECM-LSE method learns directly from the original feature space. The LSE encoding effectively learns the distinguishing characteristics of classes in a lower dimensional feature space and allows the visualization of proteins sequences. This aspect of ECM-LSE is further explained in <xref ref-type="sec" rid="s3-4">Section&#x20;3.4</xref>.</p>
</sec>
<sec id="s3-3">
<title>3.3 Verification on Experimentally Verified Human ECM Proteins</title>
<p>To verify the practical usefulness of our method, herein, we perform the validation of our method on experimentally verified ECM proteins. In particular, we collected 20 experimentally verified human ECM proteins from UniProt (<xref ref-type="bibr" rid="B8">Consortium, 2018</xref>). The collected sequences were not present in the positive or negative datasets of ECM-LSE. The criteria for the selection were based on the clear experimental evidence in the literature for the given sequence entry. We evaluated the EcmPred (<xref ref-type="bibr" rid="B28">Kandaswamy et&#x20;al., 2013</xref>), ECMSRC (<xref ref-type="bibr" rid="B45">Naseem et&#x20;al., 2017</xref>), PECM (<xref ref-type="bibr" rid="B69">Zhang et&#x20;al., 2014</xref>), and ECM-LSE methods. As shown in <xref ref-type="table" rid="T3">Table&#x20;3</xref>, ECM-LSE (<italic>k</italic>&#x20;&#x3d; 8 and <italic>LV</italic> &#x3d; 7) correctly identified 19 proteins as extracellular matrix proteins, whereas PECM, ECMSRC, and EcmPred identified 18, 16, and 15 proteins, respectively. It is noteworthy to point out that the models were trained on ECM proteins from metazoans; therefore, the superior performance of the proposed ECM-LSE on proteins from a completely different organism suggests that it can be effectively utilized for the annotation of unknown proteins.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Prediction results for 20 experimentally verified ECM proteins. &#x201c;<italic>
<bold>&#x2714;</bold>
</italic>&#x201d; indicates correctly identification while &#x201c;&#x2717;&#x201d; represents an incorrect identification.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">UniProtKB ACC</th>
<th align="left">NCBI definition</th>
<th align="center">EcmPred</th>
<th align="center">ECMSRC</th>
<th align="center">PECM</th>
<th align="center">ECM-LSE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Q9BY76</td>
<td align="left">Angiopoietin-related protein</td>
<td align="center">
<bold>
<italic>&#x2714;</italic>
</bold>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">P07355</td>
<td align="left">Annexin A2</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">Q9BXN1</td>
<td align="left">Asporin</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">P01137</td>
<td align="left">Transforming growth factor beta-1</td>
<td align="center">&#x2717;</td>
<td align="center">&#x2717;</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">Q8N6G6</td>
<td align="left">ADAMTS-like protein 1</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">P27797</td>
<td align="left">Calreticulin</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">Q76M96</td>
<td align="left">Coiled coil domain&#x2013;containing protein</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">Q07654</td>
<td align="left">Trefoil factor 3</td>
<td align="center">&#x2717;</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">&#x2717;</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">O75339</td>
<td align="left">Cartilage intermediate layer protein 1</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">Q15063</td>
<td align="left">Periostin</td>
<td align="center">&#x2717;</td>
<td align="center">&#x2717;</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">O43405</td>
<td align="left">Cochlin</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">Q96P44</td>
<td align="left">Collagen alpha-1(XXI) chain</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">P01009</td>
<td align="left">Alpha-1-antitrypsin</td>
<td align="center">&#x2717;</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">&#x2717;</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">Q14118</td>
<td align="left">Dystroglycan</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">&#x2717;</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">Q12805</td>
<td align="left">EGF-containing fibulin-like extracellular matrix protein 1</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">Q75N90</td>
<td align="left">Fibrillin-3</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">P09382</td>
<td align="left">Galectin-1</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">Q8N2S1</td>
<td align="left">Latent-transforming growth factor beta&#x2013;binding protein 4</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">P27487</td>
<td align="left">Dipeptidyl peptidase 4</td>
<td align="center">&#x2717;</td>
<td align="center">&#x2717;</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
<tr>
<td align="left">P08253</td>
<td align="left">72&#xa0;kDa type IV collagenase</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
<td align="center">
<italic>&#x2714;</italic>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-4">
<title>3.4 Discussion</title>
<p>For typical classification problems such as lysine acetylation site prediction in proteins (<xref ref-type="bibr" rid="B59">Wu et&#x20;al., 2019</xref>) or the identification of protein&#x2013;protein binding sites (<xref ref-type="bibr" rid="B12">Fernandez-Recio et&#x20;al., 2005</xref>), a large number of positive and negative samples are usually available in the datasets. Therefore, the problem of class imbalance or intra-class variation is not a major concern (<xref ref-type="bibr" rid="B20">Johnson and Khoshgoftaar, 2019</xref>). However, the limited availability of ECM samples results in an imbalanced dataset, resulting in an ill-posed problem. A number of approaches, including sample rescaling, have been proposed in the literature to tackle the imbalanced data problem (<xref ref-type="bibr" rid="B62">Xiao et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B25">Kabir et&#x20;al., 2018</xref>). Classifiers based on these rescaling techniques tend to behave well; however, the generalization of the method is compromised. Furthermore, the comparison of methods using rescaled samples with the methods using a standard dataset is not reasonable. In the proposed study, we utilize a standard dataset and develop a method that effectively discriminates the ECM proteins from non-ECM. This is achieved through the latent space learning of the CKSAAP features. For better understanding, we compare the t-SNE projection of the CKSAAP features with the proposed latent space in <xref ref-type="fig" rid="F5">Figure&#x20;5</xref>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Feature embedding <bold>(A)</bold> using t-SNE (<xref ref-type="bibr" rid="B57">Van der Maaten and Hinton, 2008)</xref> projection of the original feature space and <bold>(B)</bold> using the proposed latent space encoding (ECM-LSE) method.</p>
</caption>
<graphic xlink:href="fbioe-09-752658-g005.tif"/>
</fig>
<p>For visualization purposes, the data were projected on two dimensions using t-SNE (<xref ref-type="bibr" rid="B57">Van der Maaten and Hinton, 2008</xref>) projection of the original feature space and two variable latent spaces in the case of ECM-LSE. In the t-SNE projection shown in <xref ref-type="fig" rid="F5">Figure&#x20;5A</xref>, it can be observed that both ECMs and non-ECMs appear in an overlapping fashion, suggesting that the development of the ECM classifier using original feature space is an arduous task. As shown in <xref ref-type="fig" rid="F5">Figure&#x20;5B</xref>, the proposed latent space encoding (ECM-LSE) presents superior learning capabilities and maps the ECMs and non-ECMs in separate regions in contrast to the unsupervised subspace learning method of t-SNE (<xref ref-type="bibr" rid="B57">Van der Maaten and Hinton, 2008</xref>).</p>
<p>The proposed method, as shown in <xref ref-type="fig" rid="F5">Figure&#x20;5B</xref>, tends to form distinguishable clusters of ECM and non-ECM proteins. Although some overlap can be observed in the projection of the proposed method, it is still remarkably better than that of the t-SNE, and since the projection is shown for two latent variables only, the actual model with seven latent variables is expected to mitigate the overlap to a greater extent. These projections are also helpful in understanding the working principle of the proposed method and the motivation for the development of nonlinear auto-encoded learning of latent&#x20;space.</p>
<p>The proposed hybrid approach presents a hybrid design with capabilities of efficient feature selection and classification of ECM proteins. The latent space dynamically reduces the dimension of the feature space and retains only the relevant information sufficient to efficiently distinguish ECM from non-ECM samples. Although, the proposed method can predict ECM from different organisms, it is not a replacement for gold standard wet lab&#x2013;based testing. Furthermore, due to the scarcity of available ECM proteins the model may show biased performance in favor of already explored ECM and finding novel proteins may require the fusion of additional information. However, efforts have been made to avoid overfitting in order to seek the generalization property of the model by deploying dropout and batch normalization techniques. Further enhancements to the ECM prediction task where scarcity of the positive samples persists can be made by applying a transfer learning approach, where a large scale model is trained on a closely related dataset and is further fine-tuned for ECM samples. The Python implementation of the proposed algorithm has been made public, and interested users can utilize the algorithm for their problem of interest. The algorithm is available at (<ext-link ext-link-type="uri" xlink:href="https://github.com/Shujaat123/ECM-LSE/blob/master/ECM_LSE_Online.ipynb">https://github.com/Shujaat123/ECM-LSE/blob/master/ECM_LSE_Online.ipynb</ext-link>). In the future, we aim to explore the efficacy of the auto-encoder&#x2013;based classifiers on other bioinformatics problems.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Conclusion</title>
<p>ECM is a complex meshwork of cross-linked proteins responsible for the architectural support of cells and contributes to the functionality of the living tissue. They also contribute toward the formation of the cancer stem cells; therefore, their study and classification from non-ECMs proteins is of prime importance. A reliable prediction method can not only help understand various abnormalities associated with several cancer types but will also assist in diagnostic research. Conventional experimental-based methods are considered gold standards for this task; however, they are extremely time consuming and scanning a large number of proteins is practically infeasible. In this research, we designed a latent space learning method for the classification of ECM proteins. The proposed method can be used as a reliable prediction model. An important feature of the proposed method is its latent space-based projections through which protein sequences can be visualized in filtered and reduced dimensions, which is extremely helpful in finding useful clusters. The proposed method has been tested on a benchmark dataset and results of widely used performance metrics are reported. In particular, we report a balanced test accuracy of 86.45<italic>%</italic> with 0.71 Youden&#x2019;s index and 0.39 MCC (with <italic>k</italic>&#x20;&#x3d; 8 and <italic>LV</italic> &#x3d; 7). Additionally, the model performance is verified on completely unseen experimentally verified ECM proteins and shown to achieve highest prediction&#x20;score.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://github.com/Shujaat123/ECM-LSE">https://github.com/Shujaat123/ECM-LSE</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>All authors read the final manuscript and validated the results. Specific individual contributions are as follows: UA-S: supervision. MU: visualization and writing&#x2014;original draft preparation. IN: writing&#x2014;reviewing and editing. MN: project administration. AJ: writing&#x2014;reviewing and editing. MA: funding acquisition. HA: funding acquisition. SK: conceptualization, visualization, methodology, and investigation.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This research work was funded by the Institutional Fund Project under grant no. IFPHI-139-135-2020. Therefore, the authors gratefully acknowledge technical and financial support from the Ministry of Education and King Abdulaziz University, DSR, Jeddah, Saudi Arabia.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>The author would like to thank the UniProtKB and NCBI community for providing public database of protein sequences.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Asim</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Ibrahim</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Imran Malik</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dengel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Advances in Computational Methodologies for Classification and Sub-cellular Locality Prediction of Non-coding Rnas</article-title>. <source>Ijms</source> <volume>22</volume>, <fpage>8719</fpage>. <pub-id pub-id-type="doi">10.3390/ijms22168719</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Prediction of Protein Ubiquitination Sites in Arabidopsis Thaliana</article-title>. <source>Cbio</source> <volume>14</volume>, <fpage>614</fpage>&#x2013;<lpage>620</lpage>. <pub-id pub-id-type="doi">10.2174/1574893614666190311141647</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Q.-Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>P.-F.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Predicting Protein Lysine Phosphoglycerylation Sites by Hybridizing many Sequence Based Features</article-title>. <source>Mol. Biosyst.</source> <volume>13</volume>, <fpage>874</fpage>&#x2013;<lpage>882</lpage>. <pub-id pub-id-type="doi">10.1039/c6mb00875e</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chicco</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Jurman</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The Advantages of the matthews Correlation Coefficient (Mcc) over F1 Score and Accuracy in Binary Classification Evaluation</article-title>. <source>BMC genomics</source> <volume>21</volume>, <fpage>6</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/s12864-019-6413-7</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Some Remarks on Protein Attribute Prediction and Pseudo Amino Acid Composition</article-title>. <source>J.&#x20;Theor. Biol.</source> <volume>273</volume>, <fpage>236</fpage>&#x2013;<lpage>247</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2010.12.024</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.-C.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>iLoc-Hum: Using the Accumulation-Label Scale to Predict Subcellular Locations of Human Proteins with Both Single and Multiple Sites</article-title>. <source>Mol. Biosyst.</source> <volume>8</volume>, <fpage>629</fpage>&#x2013;<lpage>641</lpage>. <pub-id pub-id-type="doi">10.1039/c1mb05420a</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Comon</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Independent Component Analysis, a New Concept</article-title>. <source>Signal. Processing</source> <volume>36</volume>, <fpage>287</fpage>&#x2013;<lpage>314</lpage>. <pub-id pub-id-type="doi">10.1016/0165-1684(94)90029-9</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Consortium</surname>
<given-names>T. U.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>UniProt: a Worldwide Hub of Protein Knowledge</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume>, <fpage>D506</fpage>&#x2013;<lpage>D515</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1049</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Brooks Iii</surname>
<given-names>C. L.</given-names>
<suffix>III</suffix>
</name>
</person-group> (<year>2019</year>). <article-title>Deciphering Protein Evolution and Fitness Landscapes with Latent Space Models</article-title>. <source>Nat. Commun.</source> <volume>10</volume>, <fpage>5644</fpage>&#x2013;<lpage>5657</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-13633-0</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dyrl&#xf8;v Bendtsen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Nielsen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>von Heijne</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Brunak</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Improved Prediction of Signal Peptides: SignalP 3.0</article-title>. <source>J.&#x20;Mol. Biol.</source> <volume>340</volume>, <fpage>783</fpage>&#x2013;<lpage>795</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2004.05.028</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Endo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ishiwata-Endo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yamada</surname>
<given-names>K. M.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Extracellular Matrix Protein Anosmin Promotes Neural Crest Formation and Regulates Fgf, Bmp, and Wnt Activities</article-title>. <source>Developmental Cel.</source> <volume>23</volume>, <fpage>305</fpage>&#x2013;<lpage>316</lpage>. <pub-id pub-id-type="doi">10.1016/j.devcel.2012.07.006</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fernandez-Recio</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Totrov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Skorodumov</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Abagyan</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Optimal Docking Area: a New Method for Predicting Protein-Protein Interaction Sites</article-title>. <source>Proteins</source> <volume>58</volume>, <fpage>134</fpage>&#x2013;<lpage>143</lpage>. <pub-id pub-id-type="doi">10.1002/prot.20285</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gonzalez-Pujana</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Santos-Vizcaino</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Garc&#xed;a-Hernando</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hernaez-Estrada</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>M. de Pancorbo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Benito-Lopez</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Extracellular Matrix Protein Microarray-Based Biosensor with Single Cell Resolution: Integrin Profiling and Characterization of Cell-Biomaterial Interactions</article-title>. <source>Sensors Actuators B: Chem.</source> <volume>299</volume>, <fpage>126954</fpage>. <pub-id pub-id-type="doi">10.1016/j.snb.2019.126954</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hanchuan Peng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Fuhui Long</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Feature Selection Based on Mutual Information Criteria of max-dependency, max-relevance, and Min-Redundancy</article-title>. <source>IEEE Trans. Pattern Anal. Machine Intell.</source> <volume>27</volume>, <fpage>1226</fpage>&#x2013;<lpage>1238</lpage>. <pub-id pub-id-type="doi">10.1109/tpami.2005.159</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hiraki</surname>
<given-names>H. L.</given-names>
</name>
<name>
<surname>Matera</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Rose</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Kent</surname>
<given-names>R. N.</given-names>
</name>
<name>
<surname>Todd</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Stout</surname>
<given-names>M. E.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Magnetic Alignment of Electrospun Fiber Segments within a Hydrogel Composite Guides Cell Spreading and Migration Phenotype Switching</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>9</volume>, <fpage>679165</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2021.679165</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Horton</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>K.-J.</given-names>
</name>
<name>
<surname>Obayashi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Fujita</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Harada</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Adams-Collier</surname>
<given-names>C. J.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>WoLF PSORT: Protein Localization Predictor</article-title>. <source>Nucleic Acids Res.</source> <volume>35</volume>, <fpage>W585</fpage>&#x2013;<lpage>W587</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkm259</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Horton</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Obayashi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Nakai</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Protein Subcellular Localisation Prediction with WoLF PSORT</article-title>. <source>APBC</source> <volume>35</volume>, <fpage>39</fpage>&#x2013;<lpage>48</lpage>. </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>K.-c.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Ideepsubmito: Identification of Protein Submitochondrial Localization with Deep Learning</article-title>. <source>Brief. Bioinform.</source>, <fpage>bbab288</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab288</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Van den Steen</surname>
<given-names>P. E.</given-names>
</name>
<name>
<surname>Sang</surname>
<given-names>Q.-X. A.</given-names>
</name>
<name>
<surname>Opdenakker</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Matrix Metalloproteinase Inhibitors as Therapy for Inflammatory and Vascular Diseases</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>6</volume>, <fpage>480</fpage>&#x2013;<lpage>498</lpage>. <pub-id pub-id-type="doi">10.1038/nrd2308</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johnson</surname>
<given-names>J.&#x20;M.</given-names>
</name>
<name>
<surname>Khoshgoftaar</surname>
<given-names>T. M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Survey on Deep Learning with Class Imbalance</article-title>. <source>J.&#x20;Big Data</source> <volume>6</volume>, <fpage>27</fpage>. <pub-id pub-id-type="doi">10.1186/s40537-019-0192-5</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jolliffe</surname>
<given-names>I. T.</given-names>
</name>
</person-group> (<year>1986</year>). <article-title>Principal Components in Regression Analysis</article-title>. <source>Principal Component Analysis</source>. <publisher-name>Springer</publisher-name>, <fpage>129</fpage>&#x2013;<lpage>155</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-4757-1904-8_8</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ju</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.-Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Prediction of Citrullination Sites by Incorporating K-Spaced Amino Acid Pairs into Chou&#x27;s General Pseudo Amino Acid Composition</article-title>. <source>Gene</source> <volume>664</volume>, <fpage>78</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1016/j.gene.2018.04.055</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ju</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.-Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Prediction of Lysine Formylation Sites Using the Composition of K-Spaced Amino Acid Pairs via Chou&#x27;s 5-steps Rule and General Pseudo Components</article-title>. <source>Genomics</source> <volume>112</volume>, <fpage>859</fpage>&#x2013;<lpage>866</lpage>. <pub-id pub-id-type="doi">10.1016/j.ygeno.2019.05.027</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jung</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ryu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hwang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Prediction of Extracellular Matrix Proteins Based on Distinctive Sequence and Domain Characteristics</article-title>. <source>J.&#x20;Comput. Biol.</source> <volume>17</volume>, <fpage>97</fpage>&#x2013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1089/cmb.2008.0236</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kabir</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ahmad</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Iqbal</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khan Swati</surname>
<given-names>Z. N.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>D.-J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Improving Prediction of Extracellular Matrix Proteins Using Evolutionary Information via a Grey System Model and Asymmetric Under-sampling Technique</article-title>. <source>Chemometrics Intell. Lab. Syst.</source> <volume>174</volume>, <fpage>22</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1016/j.chemolab.2018.01.004</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kandaswamy</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
<name>
<surname>Martinetz</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>M&#xf6;ller</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Suganthan</surname>
<given-names>P. N.</given-names>
</name>
<name>
<surname>Sridharan</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>AFP-pred: A Random forest Approach for Predicting Antifreeze Proteins from Sequence-Derived Properties</article-title>. <source>J.&#x20;Theor. Biol.</source> <volume>270</volume>, <fpage>56</fpage>&#x2013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2010.10.037</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kandaswamy</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Pugalenthi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hartmann</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Kalies</surname>
<given-names>K.-U.</given-names>
</name>
<name>
<surname>M&#xf6;ller</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Suganthan</surname>
<given-names>P. N.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>SPRED: A Machine Learning Approach for the Identification of Classical and Non-classical Secretory Proteins in Mammalian Genomes</article-title>. <source>Biochem. Biophysical Res. Commun.</source> <volume>391</volume>, <fpage>1306</fpage>&#x2013;<lpage>1311</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbrc.2009.12.019</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kandaswamy</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Pugalenthi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kalies</surname>
<given-names>K.-U.</given-names>
</name>
<name>
<surname>Hartmann</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Martinetz</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>EcmPred: Prediction of Extracellular Matrix Proteins Based on Random forest with Maximum Relevance Minimum Redundancy Feature Selection</article-title>. <source>J.&#x20;Theor. Biol.</source> <volume>317</volume>, <fpage>377</fpage>&#x2013;<lpage>383</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2012.10.015</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karag&#xf6;z</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Geuens</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>LaPointe</surname>
<given-names>V. L.</given-names>
</name>
<name>
<surname>van Griensven</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Carlier</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Win, Lose, or Tie: Mathematical Modeling of Ligand Competition at the Cell&#x2013;Extracellular Matrix Interface</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>9</volume>, <fpage>340</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2021.657244</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Naseem</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Togneri</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bennamoun</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Rafp-pred: Robust Prediction of Antifreeze Proteins Using Localized Analysis of N-Peptide Compositions</article-title>. <source>Ieee/acm Trans. Comput. Biol. Bioinf.</source> <volume>15</volume>, <fpage>244</fpage>&#x2013;<lpage>250</lpage>. <pub-id pub-id-type="doi">10.1109/tcbb.2016.2617337</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>S.-H.</given-names>
</name>
<name>
<surname>Turnbull</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guimond</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Extracellular Matrix and Cell Signalling: the Dynamic Cooperation of Integrin, Proteoglycan and Growth Factor Receptor</article-title>. <source>J.&#x20;Endocrinol.</source> <volume>209</volume>, <fpage>139</fpage>&#x2013;<lpage>151</lpage>. <pub-id pub-id-type="doi">10.1530/joe-10-0377</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kizawa</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kou</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Iida</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sudo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Miyamoto</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fukuda</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>An Aspartic Acid Repeat Polymorphism in Asporin Inhibits Chondrogenesis and Increases Susceptibility to Osteoarthritis</article-title>. <source>Nat. Genet.</source> <volume>37</volume>, <fpage>138</fpage>&#x2013;<lpage>144</lpage>. <pub-id pub-id-type="doi">10.1038/ng1496</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klavert</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>van der Eerden</surname>
<given-names>B. C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Fibronectin in Fracture Healing: Biological Mechanisms and Regenerative Avenues</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>9</volume>, <fpage>274</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2021.663357</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klee</surname>
<given-names>E. W.</given-names>
</name>
<name>
<surname>Sosa</surname>
<given-names>C. P.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Computational Classification of Classically Secreted Proteins</article-title>. <source>Drug Discov. Today</source> <volume>12</volume>, <fpage>234</fpage>&#x2013;<lpage>240</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2007.01.008</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lewis</surname>
<given-names>D. D.</given-names>
</name>
<name>
<surname>Villarreal</surname>
<given-names>F. D.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Synthetic Biology outside the Cell: Linking Computational Tools to Cell-free Systems</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>2</volume>, <fpage>66</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2014.00066</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Prediction of Anticancer Peptides Using a Low-Dimensional Feature Model</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>8</volume>, <fpage>892</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2020.00892</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jaroszewski</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Godzik</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Clustering of Highly Homologous Sequences to Reduce the Size of Large Protein Databases</article-title>. <source>Bioinformatics</source> <volume>17</volume>, <fpage>282</fpage>&#x2013;<lpage>283</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/17.3.282</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lyu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Deepcso: a Deep-Learning Network Approach to Predicting Cysteine S-Sulphenylation Sites</article-title>. <source>Front Cel Dev Biol.</source> <volume>8</volume>, <fpage>594587</fpage>. <pub-id pub-id-type="doi">10.3389/fcell.2020.594587</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lyu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Shuai</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Protein Secondary Structure Prediction with a Reductive Deep Learning Method</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>9</volume>, <fpage>687426</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2021.687426</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Tremmel</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lietz</surname>
<given-names>C. B.</given-names>
</name>
<name>
<surname>Sackett</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Odorico</surname>
<given-names>J.&#x20;S.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>In Depth Quantification of Extracellular Matrix Proteins from Human Pancreas</article-title>. <source>J.&#x20;Proteome Res.</source> <volume>18</volume>, <fpage>3156</fpage>&#x2013;<lpage>3165</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jproteome.9b00241</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mathews</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bhonde</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>P. K.</given-names>
</name>
<name>
<surname>Totey</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Extracellular Matrix Protein Mediated Regulation of the Osteoblast Differentiation of Bone Marrow Derived Human Mesenchymal Stem Cells</article-title>. <source>Differentiation</source> <volume>84</volume>, <fpage>185</fpage>&#x2013;<lpage>192</lpage>. <pub-id pub-id-type="doi">10.1016/j.diff.2012.05.001</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McInnes</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Healy</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Melville</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Umap: Uniform Manifold Approximation and Projection for Dimension Reduction</article-title>. <comment>arXiv preprint arXiv:1802.03426</comment> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mikolov</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Corrado</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dean</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Distributed Representations of Words and Phrases and Their Compositionality</article-title>. <comment>arXiv preprint arXiv:1310.4546.</comment> </citation>
</ref>
<ref id="B44">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mitchell</surname>
<given-names>T. M.</given-names>
</name>
</person-group> (<year>1997</year>). <source>Machine Learning. 1997</source>, <volume>45</volume>. <publisher-loc>Burr Ridge, IL</publisher-loc>: <publisher-name>McGraw-Hill</publisher-name>, <fpage>870</fpage>&#x2013;<lpage>877</lpage>. </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Naseem</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Togneri</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bennamoun</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Ecmsrc: A Sparse Learning Approach for the Prediction of Extracellular Matrix Proteins</article-title>. <source>Curr. Bioinformatics</source> <volume>12</volume>, <fpage>361</fpage>&#x2013;<lpage>368</lpage>. <pub-id pub-id-type="doi">10.2174/1574893611666151215213508</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Olsen</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>Yesiltas</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Marin</surname>
<given-names>F. I.</given-names>
</name>
<name>
<surname>Pertseva</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Garc&#xed;a-Moreno</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Gregersen</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Anoxpepred: Using Deep Learning for the Prediction of Antioxidative Properties of Peptides</article-title>. <source>Sci. Rep.</source> <volume>10</volume>, <fpage>21471</fpage>&#x2013;<lpage>21481</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-78319-w</pub-id> </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Otzen</surname>
<given-names>D. E.</given-names>
</name>
<name>
<surname>Dueholm</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Najarzadeh</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Knowles</surname>
<given-names>T. P. J.</given-names>
</name>
<name>
<surname>Ruggeri</surname>
<given-names>F. S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>
<italic>In Situ</italic> Sub&#x2010;Cellular Identification of Functional Amyloids in Bacteria and Archaea by Infrared Nanospectroscopy</article-title>. <source>Small Methods</source> <volume>5</volume>, <fpage>2001002</fpage>. <pub-id pub-id-type="doi">10.1002/smtd.202001002</pub-id> </citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Moinuddin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Al-Saggaf</surname>
<given-names>U. M.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>Gssmd: A New Standardized Effect Size Measure to Improve Robustness and Interpretability in Biological Applications</article-title>. In 2020&#x20;<conf-name>IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name>, <conf-loc>Seoul, South Korea</conf-loc>, <conf-date>December 16-19, 2020</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>1096</fpage>&#x2013;<lpage>1099</lpage>. <pub-id pub-id-type="doi">10.1109/bibm49941.2020.9313582</pub-id> </citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wahab</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>E3-targetpred: Prediction of e3-target proteins using deep latent space encoding</article-title>. <comment>arXiv preprint arXiv:2007.12073</comment> </citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pratiwi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Schaduangrat</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Prachayasittikul</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Wikberg</surname>
<given-names>J.&#x20;E.</given-names>
</name>
<name>
<surname>Nantasenamat</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Cryoprotect: a Web Server for Classifying Antifreeze Proteins from Nonantifreeze Proteins</article-title>. <source>J.&#x20;Chem.</source> <volume>2017</volume>. <pub-id pub-id-type="doi">10.1155/2017/9861752</pub-id> </citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ras-Carmona</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gomez-Perosanz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Reche</surname>
<given-names>P. A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Prediction of Unconventional Protein Secretion by Exosomes</article-title>. <source>BMC bioinformatics</source> <volume>22</volume>, <fpage>333</fpage>&#x2013;<lpage>346</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-021-04219-z</pub-id> </citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sch&#xf6;lkopf</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Smola</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname>
<given-names>K.-R.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Nonlinear Component Analysis as a Kernel Eigenvalue Problem</article-title>. <source>Neural Comput.</source> <volume>10</volume>, <fpage>1299</fpage>&#x2013;<lpage>1319</lpage>. <pub-id pub-id-type="doi">10.1162/089976698300017467</pub-id> </citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Prediction of Genetic Factors of Hyperthyroidism Based on Gene Interaction Network</article-title>. <source>Front. Cel Developmental Biol.</source>, <fpage>1668</fpage>. <pub-id pub-id-type="doi">10.3389/fcell.2021.700355</pub-id> </citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Student</surname>
</name>
</person-group> (<year>1908</year>). <article-title>The Probable Error of a Mean</article-title>. <source>Biometrika</source> <volume>1&#x2013;25</volume>. <pub-id pub-id-type="doi">10.2307/2331554</pub-id> </citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Usman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.&#x20;A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Afp-lse: Antifreeze Proteins Prediction Using Latent Space Encoding of Composition of K-Spaced Amino Acid Pairs</article-title>. <source>Sci. Rep.</source> <volume>10</volume>, <fpage>7197</fpage>&#x2013;<lpage>7210</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-63259-2</pub-id> </citation>
</ref>
<ref id="B56">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Usman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.&#x20;A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Afp-cksaap: Prediction of Antifreeze Proteins Using Composition of K-Spaced Amino Acid Pairs with Deep Neural Network</article-title>. In 2019&#x20;<conf-name>IEEE 19th International Conference on Bioinformatics and Bioengineering (BIBE)</conf-name>, <conf-loc>Athens, Greece</conf-loc>, <conf-date>October 28-30, 2019</conf-date>, (<publisher-name>IEEE</publisher-name>), <fpage>38</fpage>&#x2013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1109/bibe.2019.00016</pub-id> </citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van der Maaten</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Visualizing Data Using T-Sne</article-title>. <source>J.&#x20;machine Learn. Res.</source> <volume>9</volume>. </citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Prpred: A Predictor to Identify Plant Resistance Proteins by Incorporating K-Spaced Amino Acid (Group) Pairs</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>8</volume>, <fpage>1593</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2020.645520</pub-id> </citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A Deep Learning Method to More Accurately Recall Known Lysine Acetylation Sites</article-title>. <source>BMC bioinformatics</source> <volume>20</volume>, <fpage>49</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-019-2632-9</pub-id> </citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Z.-C.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>iLoc-Gpos: A Multi-Layer Classifier for Predicting the Subcellular Localization of Singleplex and Multiplex Gram-Positive Bacterial Proteins</article-title>. <source>Ppl</source> <volume>19</volume>, <fpage>4</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.2174/092986612798472839</pub-id> </citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Z.-C.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>iLoc-Plant: A Multi-Label Classifier for Predicting the Subcellular Localization of Plant Proteins with Both Single and Multiple Sites</article-title>. <source>Mol. Biosyst.</source> <volume>7</volume>, <fpage>3287</fpage>&#x2013;<lpage>3297</lpage>. <pub-id pub-id-type="doi">10.1039/c1mb05232b</pub-id> </citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hui</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Iafp-Ense: an Ensemble Classifier for Identifying Antifreeze Protein by Incorporating Grey Model and Pssm into Pseaac</article-title>. <source>J.&#x20;Membr. Biol.</source> <volume>249</volume>, <fpage>845</fpage>&#x2013;<lpage>854</lpage>. <pub-id pub-id-type="doi">10.1007/s00232-016-9935-9</pub-id> </citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.-C.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>K.-C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>iLoc-Virus: A Multi-Label Learning Classifier for Identifying the Subcellular Localization of Virus Proteins with Both Single and Multiple Sites</article-title>. <source>J.&#x20;Theor. Biol.</source> <volume>284</volume>, <fpage>42</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2011.06.005</pub-id> </citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiaowei</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhiqiang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Minghao</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Using Support Vector Machine and Evolutionary Profiles to Predict Antifreeze Protein Sequences</article-title>. <source>Int. J.&#x20;Mol. Sci.</source> <volume>13</volume>, <fpage>2196</fpage>&#x2013;<lpage>2207</lpage>. </citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>An Effective Antifreeze Protein Predictor with Ensemble Classifiers and Comprehensive Sequence Descriptors</article-title>. <source>Ijms</source> <volume>16</volume>, <fpage>21191</fpage>&#x2013;<lpage>21214</lpage>. <pub-id pub-id-type="doi">10.3390/ijms160921191</pub-id> </citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>C.-S.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>C.-H.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Identification of Antifreeze Proteins and Their Functional Residues by Support Vector Machine and Genetic Algorithms Based on N-Peptide Compositions</article-title>. <source>PloS one</source> <volume>6</volume>, <fpage>e20445</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0020445</pub-id> </citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zemouri</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Semi-supervised Adversarial Variational Autoencoder</article-title>. <source>Make</source> <volume>2</volume>, <fpage>361</fpage>&#x2013;<lpage>378</lpage>. <pub-id pub-id-type="doi">10.3390/make2030020</pub-id> </citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.-D.</given-names>
</name>
<name>
<surname>Zulfiqar</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>S.-S.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.-L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.-Y.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Iblp: An Xgboost-Based Predictor for Identifying Bioluminescent Proteins</article-title>. <source>Comput. Math. Methods Med.</source> <volume>2021</volume>, <fpage>6664362</fpage>. <pub-id pub-id-type="doi">10.1155/2021/6664362</pub-id> </citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>PECM: Prediction of Extracellular Matrix Proteins Using the Concept of Chou&#x27;s Pseudo Amino Acid Composition</article-title>. <source>J.&#x20;Theor. Biol.</source> <volume>363</volume>, <fpage>412</fpage>&#x2013;<lpage>418</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtbi.2014.08.002</pub-id> </citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zang</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Prediction and Collection of Protein&#x2013;Metabolite Interactions</article-title>. <source>Brief. Bioinform.</source> <volume>22</volume>, <fpage>bbab014</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab014</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>