<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">842127</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.842127</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>SSH2.0: A Better Tool for Predicting the Hydrophobic Interaction Risk of Monoclonal Antibody</article-title>
<alt-title alt-title-type="left-running-head">Zhou et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">Antibody Hydrophobic Interaction Risk Prediction</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhou</surname>
<given-names>Yuwei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xie</surname>
<given-names>Shiyang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Yue</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jiang</surname>
<given-names>Lixu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Siqi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1618891/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Wei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Abagna</surname>
<given-names>Hamza Bukari</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Ning</surname>
<given-names>Lin</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1586656/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Huang</surname>
<given-names>Jian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/449350/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Center for Informational Biology</institution>, <institution>University of Electronic Science and Technology of China</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Life Science and Technology</institution>, <institution>University of Electronic Science and Technology of China</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Healthcare Technology</institution>, <institution>Chengdu Neusoft University</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1385377/overview">Chuan Dong</ext-link>, Wuhan University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/875818/overview">Jin-Xing Liu</ext-link>, Qufu Normal University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1619513/overview">Chengchi Fang</ext-link>, Institute of Hydrobiology (CAS), China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Lin Ning, <email>NingLin@nsu.edu.cn</email>; Jian Huang, <email>hj@uestc.edu.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>03</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>842127</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>12</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>01</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Zhou, Xie, Yang, Jiang, Liu, Li, Abagna, Ning and Huang.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Zhou, Xie, Yang, Jiang, Liu, Li, Abagna, Ning and Huang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>Therapeutic antibodies play a crucial role in the treatment of various diseases. However, the success rate of antibody drug development is low partially because of unfavourable biophysical properties of antibody drug candidates such as the high aggregation tendency, which is mainly driven by hydrophobic interactions of antibody molecules. Therefore, early screening of the risk of hydrophobic interaction of antibody drug candidates is crucial. Experimental screening is laborious, time-consuming, and costly, warranting the development of efficient and high-throughput computational tools for prediction of hydrophobic interactions of therapeutic antibodies. In the present study, 131 antibodies with hydrophobic interaction experiment data were used to train a new support vector machine-based ensemble model, termed SSH2.0, to predict the hydrophobic interactions of antibodies. Feature selection was performed against CKSAAGP by using the graph-based algorithm MRMD2.0. Based on the antibody sequence, SSH2.0 achieved the sensitivity and accuracy of 100.00 and 83.97%, respectively. This approach eliminates the need of three-dimensional structure of antibodies and enables rapid screening of therapeutic antibody candidates in the early developmental stage, thereby saving time and cost. In addition, a web server was constructed that is freely available at <ext-link ext-link-type="uri" xlink:href="http://i.uestc.edu.cn/SSH2/">http://i.uestc.edu.cn/SSH2/</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>therapeutic antibody</kwd>
<kwd>developability</kwd>
<kwd>hydrophobic interactions</kwd>
<kwd>support vector machine</kwd>
<kwd>prediction model</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Antibodies play an indispensable role in the vertebrate immune defence system (<xref ref-type="bibr" rid="B15">Kapingidza et&#x20;al., 2020</xref>). They also serve as essential agents in biomedical research and clinical diagnostic assays such as enzyme-linked immunosorbent assay, immunohistochemical assay, and immunoprecipitation assay. Furthermore, antibodies have been extensively used in clinical treatment of many types of cancers, autoimmune diseases, and infectious diseases including the coronavirus disease 2019, which is caused by the severe acute respiratory syndrome coronavirus 2 (<xref ref-type="bibr" rid="B26">Ning et&#x20;al., 2021</xref>). Rapid development of the monoclonal antibody (mAb) technology has revolutionised pharmaceutical science and industry. Many proteins that cannot interact with small chemical molecules or are undruggable due to self-tolerance are considered efficient targets for antibody drugs. More than 550 therapeutic mAbs have been tested in phase I/II clinical trials worldwide, of which 79&#xa0;mAbs have entered the final stage of development (<xref ref-type="bibr" rid="B16">Kaplon et&#x20;al., 2020</xref>). Antibody drugs account for a large market share in the pharmaceutical industry. In 2018, the therapeutic antibodies had a global value of United&#x20;States $115.2&#xa0;billion, which is expected to reach $300 billion by the end of 2025 (<xref ref-type="bibr" rid="B23">Lu et&#x20;al., 2020</xref>). Moreover, the large-scale application of antibody phage display, single B-cell antibody, and next-generation sequencing technologies has resulted in the development of tens of thousands of preclinical therapeutic antibody drug candidates. However, the probability of a human or humanised antibody drug candidate, which is under clinical trials, being approved is low (approximately 15%) (<xref ref-type="bibr" rid="B1">Carter and Lazar, 2018</xref>). Many mAbs fail due to unfavourable physicochemical properties such as high viscosity, increased aggregation tendency, and susceptibility to chemical degradation (<xref ref-type="bibr" rid="B13">Jain et&#x20;al., 2017b</xref>).</p>
<p>Protein aggregation has been considered as one of the major challenges in biological drug development. It poses challenges during different developmental processes from fermentation and purification to storage (<xref ref-type="bibr" rid="B27">Obrezanova et&#x20;al., 2015</xref>). It not only reduces the effectiveness of a drug but also induces adverse immune responses in patients (<xref ref-type="bibr" rid="B25">Martinez Morales et&#x20;al., 2019</xref>). Thus, identifying therapeutic antibody candidates with high aggregation tendency at the early developmental stage is essential. The factors that affect protein aggregation are either intrinsic (e.g., interaction between hydrophobic patches, van der Waals forces and electrostatic interactions) or extrinsic (e.g., pH, salt concentration, buffer type, and storage conditions). Among these factors, the presence of hydrophobic moieties on the protein surface is the strongest determinant (<xref ref-type="bibr" rid="B11">Hebditch et&#x20;al., 2019</xref>). A few tools to predict the hydrophobicity of proteins including mAbs have been reported (<xref ref-type="bibr" rid="B22">Lienqueo et&#x20;al., 2006</xref>; <xref ref-type="bibr" rid="B24">Mahn et&#x20;al., 2009</xref>; <xref ref-type="bibr" rid="B7">Hanke et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B12">Jain et&#x20;al., 2017a</xref>). However, most of these tools rely on protein structures and do not provide free web services. In our previous study, we developed a tool called SSH, which can predict the hydrophobic interaction risk of mAbs solely by using the mAb sequences (<xref ref-type="bibr" rid="B5">Dzisoo et&#x20;al., 2020</xref>). The SSH tool was trained with the tripeptide composition (TPC), and the prediction accuracy of 91.226% was achieved through the voting strategy. However, the number of features used to build the SSH model is extremely higher than the number of its samples, causing concerns with overfitting and weak generalisation.</p>
<p>In the present study, we combined the experimental assay data to construct a novel in&#x20;silico tool called SSH2.0 for the prediction of hydrophobic interaction risk of mAbs. The tool developed in this study predicted hydrophobic interaction risk of mAbs by using only the amino acid sequence. Compared with the previous version, SSH2.0 was trained with new features that were optimised using a new feature selection method. Overall, SSH2.0 was superior to the previous version in terms of performance.</p>
</sec>
<sec id="s2">
<title>Dataset and Method</title>
<sec id="s2-1">
<title>Dataset</title>
<p>The antibody dataset used in a study by <xref ref-type="bibr" rid="B13">Jain et&#x20;al. (2017b)</xref> was selected in the present study. We linked the variable region in the form of &#x201c;heavy chain&#x2212;light chain&#x201d; as the antibody sequences. The dataset comprised 137 antibody sequences (48 from approved antibodies and 89 from clinical II/III trials) and data of 12 biophysical and binding assays. Six antibody sequences with conflicting records were eliminated, resulting in inclusion of 131 antibody sequences. The assays, namely stand-up monolayer adsorption chromatography (SMAC), salt-gradient affinity-capture self-interaction nanoparticle spectroscopy (SGAC-SINS), and hydrophobic interaction chromatography (HIC), were used to determine the risk of hydrophobic interaction. A threshold of 10% was employed according to a study by <xref ref-type="bibr" rid="B13">Jain et&#x20;al. (2017b)</xref> (<xref ref-type="table" rid="T1">Table&#x20;1</xref>). The antibody was labelled with a fault flag if one of the aforementioned three assay values exceeded the set threshold. We obtained 94 negative samples (0 flag) and 37 positive samples (25 with one flag, 8 with two flags, and four antibodies with exactly three flags). <xref ref-type="fig" rid="F1">Figure&#x20;1</xref> shows the detailed labelling of each antibody. To solve the problem of the dataset imbalance, 94 negative samples were randomly divided into three groups, with each group containing 31, 31, and 32 antibodies. Each sub-dataset (Group 1, Group 2, Group 3) was combined with positive samples to train three sub-models (SSH_a,SSH_b,SSH_c). Then, the results of the three sub-models was integrated, and an ensemble predictor was constructed using a voting strategy.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Three experimental thresholds for evaluating the hydrophobic interaction of antibodies (<xref ref-type="bibr" rid="B13">Jain et&#x20;al., 2017b</xref>).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Assay</th>
<th align="center">Worst 10% threshold</th>
<th align="center">Units (flag)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Standup monolayer adsorption chromatography (SMAC)</td>
<td align="char" char=".">12.8</td>
<td align="center">Retention time (min) (&#x3e;)</td>
</tr>
<tr>
<td align="left">Salt-gradient affinity-capture self-interaction nanoparticle spectroscopy (SGAC-SINS)</td>
<td align="char" char=".">370</td>
<td align="center">Salt concentration (mM) (&#x3c;)</td>
</tr>
<tr>
<td align="left">Hydrophobic interaction chromatography (HIC)</td>
<td align="char" char=".">11.7</td>
<td align="center">Retention time (min) (&#x3e;)</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The number of hydrophobic interaction flags and the classification of antibodies.</p>
</caption>
<graphic xlink:href="fgene-13-842127-g001.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>Feature Extraction and Selection</title>
<p>To construct an efficient prediction tool, appropriate feature extraction methods for transforming sequence data into numerical expressions (ideally, without distortion), in addition to a reliable benchmark data set, are crucial. Features based on sequence information such as the amino acid composition and pseudo amino acid components (<xref ref-type="bibr" rid="B9">He et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B5">Dzisoo et&#x20;al., 2020</xref>; <xref ref-type="bibr" rid="B30">Wang et&#x20;al., 2020</xref>), displayed good performance in protein and peptide classification (<xref ref-type="bibr" rid="B8">He et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B21">Li et&#x20;al., 2017</xref>; <xref ref-type="bibr" rid="B14">Kang et&#x20;al., 2019</xref>). Based on a large number of experimental results, the CKSAAGP (composition of k-spaced amino acid group pairs) (<xref ref-type="bibr" rid="B3">Chen et&#x20;al., 2009</xref>; <xref ref-type="bibr" rid="B4">Chen et&#x20;al., 2018</xref>) demonstrated the best performance in the present study. In the CKSAAGP encoding scheme, 20&#xa0;amino acids were divided into the following five groups according to their physicochemical properties: g1: aliphatic group (GAVLMI); g2: aromatic group (FYW); g3: positive charge group (KRH); g4: negative charged group (DE); g5: uncharged group (STCPNQ) (<xref ref-type="bibr" rid="B4">Chen et&#x20;al., 2018</xref>). Then, the frequency of amino acid group pairs separated by k residues was calculated (the default maximum value of k was set as 5). CKSAAGP can be defined as follows:<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>3</mml:mn>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>5</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>5</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>5</mml:mn>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of times that the composition of the residue pair <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>g</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> is separated by 0&#xa0;amino acids in the whole protein sequence; <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the total number of k-spaced amino acid pairs. For a protein of length P, k &#x3d; 0, 1, 2, 3, 4, and 5, and the values of <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are P-1, P-2, P-3, P-4, P-5, and P-6, respectively. CKSAAGP can be used to encode unequal length sequences.</p>
<p>To compare the influence of different feature extraction algorithms, we used 19 feature extraction methods on the same dataset and constructed 19 models. The feature extraction methods tested in this study are AAC, DPC, TPC, CKSAAP, DDE, GAAC, GDPC, GTPC, Moran, Geary, NMBroto, CTDC, CTDT, CTDD, CTriad, KSCTriad, SOCNumber, QSOrder, and PAAC. All feature extraction processes were performed using the iFeature (<xref ref-type="bibr" rid="B4">Chen et&#x20;al., 2018</xref>) python package, which can be obtained from github (<ext-link ext-link-type="uri" xlink:href="https://github.com/Superzchen/iFeature/">https://github.com/Superzchen/iFeature/</ext-link>).</p>
<p>High-dimensional small sample data usually cause the problem such as overfitting, longer training time and redundant features. In this study, an integrated method MRMD2.0 developed by <xref ref-type="bibr" rid="B10">He et&#x20;al. (2020)</xref> was used for feature sorting and dimension reduction. MRMD2.0 represents different feature ranking with directed graph. Then the PageRank algorithm was used to obtain the new ranking. Finally, sequential forward selection (SFS) was used to select the optimal feature subset.</p>
</sec>
<sec id="s2-3">
<title>Support Vector Machine Model Establishment</title>
<p>Owing to a high prediction accuracy and simple parameter optimisation, support vector machine (SVM) has been applied extensively in many fields such as protein&#x2212;protein interactions (<xref ref-type="bibr" rid="B29">Romero-Molina et&#x20;al., 2019</xref>), drug discovery (<xref ref-type="bibr" rid="B28">Patel et&#x20;al., 2020</xref>), and medical image processing (<xref ref-type="bibr" rid="B32">Yang et&#x20;al., 2019</xref>). The basic idea of SVM is to determine the hyperplane with the largest interval in the space, which can divide positive and negative samples effectively and accurately. We employed LIBSVM (<xref ref-type="bibr" rid="B2">Chang and Lin., 2011</xref>) to construct the SVM sub-models. Among the given four kernel functions, we chose the radial basis function (RBF) kernel to obtain the optimal kernel parameter &#x3b3; and penalty parameter <italic>C</italic>. Three sub-models were integrated through the voting strategy. The results of the three sub-models were integrated, and an antibody was predicted to have high risk of hydrophobic interaction if it was predicted as a positive sample by at least two models.</p>
</sec>
<sec id="s2-4">
<title>Performance Evaluation</title>
<p>Leave-one-out cross-validation (LOOCV) was adopted to assess the performance of each sub-model. One sample in the sub-dataset was used as the test set, whereas the remaining samples constituted the training set. This process was repeated N times (where N is the number of samples). Eventually, the average prediction accuracy was considered as the final accuracy of the sub-model. The performance of the prediction models was evaluated using the common indicators, namely sensitivity (Sn), specificity (Sp), accuracy (ACC), and Matthews correlation coefficient (MCC). MCC is a relatively balanced indicator for prediction that is mainly used to measure dichotomy. It comprehensively considers TP, TN, FP, and FN, which can avoid sample imbalance deviation. These indicators can be expressed as follows:<disp-formula id="equ2">
<mml:math id="m6">
<mml:mrow>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ3">
<mml:math id="m7">
<mml:mrow>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ4">
<mml:math id="m8">
<mml:mrow>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ5">
<mml:math id="m9">
<mml:mrow>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>where TP and TN represent the number of positive data and negative data, respectively, that were predicted correctly, whereas FP and FN represent the number of positive data and negative data, respectively, that were erroneously predicted. In addition, AUC (area under the ROC curve) was used to illustrate the performance of the model. ROC curve is a TPR vs FPR plot that illustrates the diagnostic ability of a binary classifier system as its discrimination threshold is varied. AUC value ranges from 0 to 1. A model whose prediction efficiency is 100% has an AUC value of&#x20;1.</p>
</sec>
<sec id="s2-5">
<title>Developability Index (DI) Calculation</title>
<p>The developability index (DI) of each antibody in a study by <xref ref-type="bibr" rid="B13">Jain et&#x20;al. (2017b)</xref> was computed using BIOVIA Discovery Studio 2019 (BIOINFORMATICS SOCIETY OF SICHUAN PROVINCE) with the default parameters pH &#x3d; 6 and <italic>&#x3b2;</italic> &#x3d; 0.05. The crystal structure of each antibody, if available, was downloaded from the PDB database. For the antibodies whose crystal structure was not available, we performed homology modelling to build their structure. Spearman rank correlation was used to explore the correlation between DI and 12 experiment assays (<xref ref-type="bibr" rid="B13">Jain et&#x20;al., 2017b</xref>). Statistical analysis was performed with&#x20;R4.1.0.</p>
</sec>
<sec id="s2-6">
<title>Online Web Service</title>
<p>To facilitate the use of researchers, a user-friendly web server was developed. We used HTML, CSS, PHP, JavaScript to write the interface script for web service. The data processing process script was written using <italic>Python</italic>.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec id="s3-1">
<title>Feature Selection Based on CKSAAGP</title>
<p>From a total of 150 features, the optimal feature was selected using MRMD2.0. Finally, the three sub-datasets were respectively composed of 29, 31, and 35 features. <xref ref-type="fig" rid="F2">Figure&#x20;2</xref> shows the variation of ACC with feature number during the sequential forward selection process. After feature selection, AUC was increased by at least 12% (Group 3) compared with the previous value. The prediction accuracy of the model increased with a decrease in the number of features. The small number of features also reduced the computational cost, model complexity, and the risk of overfitting. The feature dimensions of the sub-datasets were all reduced by more than 70%, which demonstrated that the performance of MRMD2.0 was excellent.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The ACC of different feature numbers during the sequential forward selection process of three sub-datasets (Group 1, Group 2, Group 3).</p>
</caption>
<graphic xlink:href="fgene-13-842127-g002.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>Model Evaluation</title>
<p>We trained three SVM sub-models based on LOOCV using the optimal features. As shown in <xref ref-type="table" rid="T2">Table&#x20;2</xref>, the accuracy rates of SSH_a, SSH_b and SSH_c for the prediction of antibody hydrophobic interaction were 80.88, 77.94 and 75.36% respectively. By considering all samples as input of each sub-model, we obtained three prediction results. To visually demonstrate the ability of each sub-model to predict the hydrophobic interaction, a receiver operating characteristics (ROC) curve was drawn (<xref ref-type="fig" rid="F3">Figure&#x20;3</xref>). The AUC value of SSH_a, SSH_b and SSH_c reached 0.8583, 0.8956, and 0.8726, respectively. According to the aforementioned analysis, an ensemble model called SSH2.0 was constructed based on voting strategy. The sensitivity of the ensemble model was 100.00%, indicating that SSH2.0 can correctly identify all antibodies with a risk of hydrophobic interaction (<xref ref-type="table" rid="T2">Table&#x20;2</xref>).</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>The prediction performance of three sub-models evaluated through leave-one-out cross-validation and that of the ensemble model evaluated through voting strategy.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Model</th>
<th align="center">Sn(%)</th>
<th align="center">Sp (%)</th>
<th align="center">ACC(%)</th>
<th align="center">MCC</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">SSH_a</td>
<td align="char" char=".">81.08</td>
<td align="char" char=".">80.64</td>
<td align="char" char=".">80.88</td>
<td align="char" char=".">0.6159</td>
<td align="char" char=".">0.8086</td>
</tr>
<tr>
<td align="left">SSH_b</td>
<td align="char" char=".">81.08</td>
<td align="char" char=".">74.19</td>
<td align="char" char=".">77.94</td>
<td align="char" char=".">0.5544</td>
<td align="char" char=".">0.7763</td>
</tr>
<tr>
<td align="left">SSH_c</td>
<td align="char" char=".">78.37</td>
<td align="char" char=".">71.87</td>
<td align="char" char=".">75.36</td>
<td align="char" char=".">0.5038</td>
<td align="char" char=".">0.7513</td>
</tr>
<tr>
<td align="left">SSH2.0</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">77.66</td>
<td align="char" char=".">83.97</td>
<td align="char" char=".">0.7039</td>
<td align="char" char=".">0.8883</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The ROC curves of three sub-models for predicting all 131 antibodies.</p>
</caption>
<graphic xlink:href="fgene-13-842127-g003.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>Comparison of Different Feature Extraction Methods</title>
<p>To comprehensively evaluate the effect of the CKSAAGP algorithm, we compared it with the other 19 feature extraction algorithms. <xref ref-type="fig" rid="F4">Figure&#x20;4</xref> shows the feature dimension and dimension decline percentage obtained using all 20 algorithms after the reduction of MRMD2.0. The dimensions of multiple methods were reduced by more than 70%; however, the number of features varied among the three sub-datasets. For example, the number of TPC features decreased from 8,000 to 71 and 75 in Group 1 and Group 2, respectively, whereas that in Group 3 was 231. These results indicated that all feature extraction algorithms were affected by the samples, whereas CKSAAGP had smaller feature dimensions in all three sub-datasets with smaller variance, which was relatively robust. Furthermore, we assessed the ensemble model based on all 20 algorithms. As shown in <xref ref-type="table" rid="T3">Table&#x20;3</xref>, although the sensitivity of multiple features had reached 100%, CKSAAGP showed the highest specificity, accuracy, MCC and AUC of 77.66%, 83.97%, 0.7093, and 0.8883, respectively. Taken together, CKSAAGP was the most proper feature type for this problem, considering feature dimensions and the performance of sub-models and ensemble&#x20;model.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Analysis of MEMD2.0 dimensionality reduction results. <bold>(A)</bold> The reduced ratio and <bold>(B)</bold> the number of features in the dimension of three sub-datasets. The numbers in parentheses are the original feature numbers of various feature extraction algorithm.</p>
</caption>
<graphic xlink:href="fgene-13-842127-g004.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The prediction performance of the ensemble model based on 20 feature extraction algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Feature</th>
<th align="center">Sn (%)</th>
<th align="center">Sp(%)</th>
<th align="center">ACC(%)</th>
<th align="center">MCC</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">CKSAAGP</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">77.66</td>
<td align="char" char=".">83.97</td>
<td align="char" char=".">0.7039</td>
<td align="char" char=".">0.8883</td>
</tr>
<tr>
<td align="left">CTriad</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">75.53</td>
<td align="char" char=".">82.44</td>
<td align="char" char=".">0.6825</td>
<td align="char" char=".">0.8777</td>
</tr>
<tr>
<td align="left">DPC</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">72.34</td>
<td align="char" char=".">80.15</td>
<td align="char" char=".">0.6518</td>
<td align="char" char=".">0.8617</td>
</tr>
<tr>
<td align="left">TPC</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">71.28</td>
<td align="char" char=".">79.39</td>
<td align="char" char=".">0.6419</td>
<td align="char" char=".">0.8564</td>
</tr>
<tr>
<td align="left">AAC</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">70.21</td>
<td align="char" char=".">78.63</td>
<td align="char" char=".">0.6322</td>
<td align="char" char=".">0.8511</td>
</tr>
<tr>
<td align="left">CKSAAP</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">69.15</td>
<td align="char" char=".">77.86</td>
<td align="char" char=".">0.6226</td>
<td align="char" char=".">0.8457</td>
</tr>
<tr>
<td align="left">NMBroto</td>
<td align="char" char=".">97.30</td>
<td align="char" char=".">69.15</td>
<td align="char" char=".">77.10</td>
<td align="char" char=".">0.5983</td>
<td align="char" char=".">0.8322</td>
</tr>
<tr>
<td align="left">DDE</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">65.96</td>
<td align="char" char=".">75.57</td>
<td align="char" char=".">0.5947</td>
<td align="char" char=".">0.8298</td>
</tr>
<tr>
<td align="left">GTPC</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">63.83</td>
<td align="char" char=".">74.05</td>
<td align="char" char=".">0.5767</td>
<td align="char" char=".">0.8191</td>
</tr>
<tr>
<td align="left">CTDC</td>
<td align="char" char=".">97.30</td>
<td align="char" char=".">65.96</td>
<td align="char" char=".">74.81</td>
<td align="char" char=".">0.5699</td>
<td align="char" char=".">0.8163</td>
</tr>
<tr>
<td align="left">CTDT</td>
<td align="char" char=".">91.89</td>
<td align="char" char=".">63.83</td>
<td align="char" char=".">71.76</td>
<td align="char" char=".">0.5021</td>
<td align="char" char=".">0.7786</td>
</tr>
<tr>
<td align="left">CTDD</td>
<td align="char" char=".">97.30</td>
<td align="char" char=".">56.38</td>
<td align="char" char=".">67.94</td>
<td align="char" char=".">0.4910</td>
<td align="char" char=".">0.7684</td>
</tr>
<tr>
<td align="left">Geary</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">53.19</td>
<td align="char" char=".">66.41</td>
<td align="char" char=".">0.4929</td>
<td align="char" char=".">0.7660</td>
</tr>
<tr>
<td align="left">SOCNumber</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">52.13</td>
<td align="char" char=".">65.65</td>
<td align="char" char=".">0.4850</td>
<td align="char" char=".">0.7606</td>
</tr>
<tr>
<td align="left">Moran</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">50.00</td>
<td align="char" char=".">64.12</td>
<td align="char" char=".">0.4693</td>
<td align="char" char=".">0.7500</td>
</tr>
<tr>
<td align="left">QSOrder</td>
<td align="char" char=".">83.78</td>
<td align="char" char=".">60.64</td>
<td align="char" char=".">67.18</td>
<td align="char" char=".">0.4003</td>
<td align="char" char=".">0.7221</td>
</tr>
<tr>
<td align="left">KSCTriad</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">40.43</td>
<td align="char" char=".">57.25</td>
<td align="char" char=".">0.4010</td>
<td align="char" char=".">0.7021</td>
</tr>
<tr>
<td align="left">GAAC</td>
<td align="char" char=".">75.68</td>
<td align="char" char=".">62.77</td>
<td align="char" char=".">66.41</td>
<td align="char" char=".">0.3464</td>
<td align="char" char=".">0.6922</td>
</tr>
<tr>
<td align="left">GDPC</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">30.85</td>
<td align="char" char=".">50.38</td>
<td align="char" char=".">0.3345</td>
<td align="char" char=".">0.6543</td>
</tr>
<tr>
<td align="left">PAAC</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">0.00</td>
<td align="char" char=".">28.24</td>
<td align="char" char=".">0.0000</td>
<td align="char" char=".">0.5000</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-4">
<title>CKSAAGP Features That Closely Related to the Hydrophobic Interaction</title>
<p>The properties of amino acid side chains are closely related to the structure and function of proteins. The nonpolar amino acids (aliphatic, and aromatic amino acids) are usually hydrophobic. Conversely, the polar amino acids (positively and negatively charged and uncharged amino acids) are hydrophilic. Among all the features in models, aliphatic. aliphatic.gap5, aromatic. aliphatic.gap3, negativecharger. aliphatic.gap1 were present in all sub-models, and only one of these features, namely aromatic. aliphatic.gap3, was in the top 10 features (<xref ref-type="table" rid="T4">Table&#x20;4</xref>). The binding of nonpolar amino acids with strong hydrophobicity increases the hydrophobicity of the protein. Interestingly, as shown in <xref ref-type="table" rid="T4">Table&#x20;4</xref>, the combination &#x201c;polar &#x2b; nonpolar&#x201d; appeared frequently, which indicated that a polar amino acid and a nonpolar amino acid are separated by several amino acids in space that probably enhances the hydrophobicity of the protein, although a single polar amino acid is hydrophilic. In summary, if the CKSAAGP features listed in <xref ref-type="table" rid="T4">Table&#x20;4</xref> appear frequently in an antibody sequence, the antibody should be excluded from early development.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>The top 10 CKSAAGP features of three sub-models. The features marked in red indicate that they exist in at least two sub-models (neg: negative charged group; pos: positive charge group).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">SSH_a</th>
<th align="center">SSH_b</th>
<th align="center">SSH_c</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">aromatic.uncharge.gap0</td>
<td align="left">aromatic.aliphatic.gap1</td>
<td align="left">aliphatic.pos.gap0</td>
</tr>
<tr>
<td align="left">uncharge.uncharge.gap0</td>
<td align="left">aliphatic.neg.gap3</td>
<td align="left">uncharge.aliphatic.gap4</td>
</tr>
<tr>
<td align="left">aromatic.aliphatic.gap3</td>
<td align="left">pos.aliphatic.gap2</td>
<td align="left">uncharge.aromatic.gap2</td>
</tr>
<tr>
<td align="left">pos.neg.gap0</td>
<td align="left">uncharge.uncharge.gap2</td>
<td align="left">neg.aromatic.gap5</td>
</tr>
<tr>
<td align="left">aliphatic.aromatic.gap5</td>
<td align="left">aliphatic.pos.gap0</td>
<td align="left">pos.uncharge.gap5</td>
</tr>
<tr>
<td align="left">uncharge.uncharge.gap2</td>
<td align="left">neg.uncharge.gap4</td>
<td align="left">aliphatic.uncharge.gap5</td>
</tr>
<tr>
<td align="left">pos.uncharge.gap0</td>
<td align="left">aliphatic.aromatic.gap5</td>
<td align="left">aromatic.aliphatic.gap3</td>
</tr>
<tr>
<td align="left">pos.uncharge.gap4</td>
<td align="left">neg.aliphatic.gap2</td>
<td align="left">aliphatic.uncharge.gap1</td>
</tr>
<tr>
<td align="left">neg.pos.gap2</td>
<td align="left">aromatic.uncharge.gap2</td>
<td align="left">aliphatic.aliphatic.gap2</td>
</tr>
<tr>
<td align="left">aliphatic.uncharge.gap5</td>
<td align="left">aromatic.pos.gap1</td>
<td align="left">neg.neg.gap3</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-5">
<title>Comparison Between the Previously Constructed SSH Model and DI Computational Tool</title>
<p>In our previous study, <xref ref-type="bibr" rid="B5">Dzisoo et&#x20;al. (2020)</xref> provided a web-server named SSH based on TPC features to predict the hydrophobic interaction risk of mAbs. However, the number of features in SSH was far more than the number of samples, which indicated the probability of overfitting. In this study, we optimized the feature extraction algorithm and feature selection method to maintain the prediction accuracy with fewer features. We uniformly defined sensitivity as the ability to identify samples with hydrophobic interaction risk. As shown in <xref ref-type="table" rid="T5">Table&#x20;5</xref>, the number of each SSH sub-model features was more than 300, whereas the number of samples used for training was &#x3c; 70. After using the CKSAAGP feature scheme and MRMD2.0 feature selection algorithm, the number of features in SSH2.0 reduced to one-tenth that of SSH. Although the ACC and AUC of the ensemble model decreased by 7.26% and 0.0737, respectively, we paid more attention to the performance to identify defective samples. The sensitivity of SSH2.0 reached 100.00%, which was 16.70% higher than that of&#x20;SSH.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Comparison of the feature and performance between SSH2.0 and SSH.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Model</th>
<th align="center">Feature</th>
<th align="center">Feature extraction method</th>
<th align="center">Feature number of sub-models</th>
<th align="center">Sn(%)</th>
<th align="center">Sp(%)</th>
<th align="center">ACC(%)</th>
<th align="center">AUC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">SSH</td>
<td align="center">TPC</td>
<td align="center">f -scores</td>
<td align="center">313,315,315</td>
<td align="char" char=".">84.30</td>
<td align="char" char=".">96.39</td>
<td align="char" char=".">91.23</td>
<td align="char" char=".">0.9620</td>
</tr>
<tr>
<td align="left">SSH2.0</td>
<td align="center">CKSAAGP</td>
<td align="center">MRMD2.0</td>
<td align="center">29,31,35</td>
<td align="char" char=".">100.00</td>
<td align="char" char=".">77.66</td>
<td align="char" char=".">83.97</td>
<td align="char" char=".">0.8883</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>DI is another widely employed tool for assessing the aggregation propensity of proteins (<xref ref-type="bibr" rid="B18">Lauer et&#x20;al., 2012</xref>). We performed the Spearman rank correlation test to explore the correlation between DI and 12 experimental assays. Surprisingly, the three most relevant assays were SMAC, SGAC-SINS and HIC (<xref ref-type="fig" rid="F5">Figure&#x20;5</xref>), which we used to assess the hydrophobic interaction risk of mAbs in the current study. The result confirmed that protein aggregation is mainly driven by hydrophobic interactions (<xref ref-type="bibr" rid="B11">Hebditch et&#x20;al., 2019</xref>). According to the methods based on the experimental data presented by <xref ref-type="bibr" rid="B13">Jain et&#x20;al. (2017b)</xref>, 37 antibodies were flagged with hydrophobic interaction warnings. We used this as the gold standard. Because high DI values correspond to low developability (<xref ref-type="bibr" rid="B18">Lauer et&#x20;al., 2012</xref>), we sorted all the antibodies according to the descending order of their DI values. The top 37 antibodies with high DI values were predicted to have the hydrophobic interaction risk. However, the prediction performance of the DI method was inferior to that of SSH2.0. The accuracy rates of SSH2.0 and DI were 83.97 and 61.83%, respectively. The results suggest that owing to the low prediction accuracy, the application of DI to a screening platform would lead to many antibodies with a high aggregation risk being incorrectly selected.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Correlation coefficient matrix of DI and 12 experimental assays. The lower triangle shows the spearman correlation coefficients, and the upper triangle represents the corresponding correlation values. The radius of the circles is proportional to the magnitude of the correlation coefficient. Red represents a positive correlation, and blue represents a negative correlation.</p>
</caption>
<graphic xlink:href="fgene-13-842127-g005.tif"/>
</fig>
</sec>
<sec id="s3-6">
<title>Web-Server Guidance</title>
<p>To serve the relevant researchers, we established a user-friendly web server for the prediction of hydrophobic interaction risk of mAbs. The server is freely accessible at <ext-link ext-link-type="uri" xlink:href="http://i.uestc.edu.cn/SSH2/">http://i.uestc.edu.cn/SSH2/</ext-link>. The homepage of SSH2.0 is shown in <xref ref-type="fig" rid="F6">Figure&#x20;6A</xref>. The variable region sequences of heavy chains and light chains were input separately. Because some antibodies only have one chain, the input consisting of single heavy or light chain were allowed. The submitted antibody sequences were in the FASTA format. The AbRSA tool can help in antibody numbering and CDR (complementarity-determining region) delimiting (<xref ref-type="bibr" rid="B20">Li et&#x20;al., 2019</xref>). SSH2.0 allowed the detection of illegal characters, and only 20 common amino acids were found to be legal for sequence input. Illegal characters such as B, J, O, U, X, Z and the numbers 1&#x2013;9 were forbidden (<xref ref-type="fig" rid="F6">Figure&#x20;6B</xref>). <xref ref-type="fig" rid="F6">Figure&#x20;6C</xref> shows the prediction results.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Screenshots of the SSH2.0 web server. <bold>(A)</bold>Homepage of the SSH2.0 web server. <bold>(B)</bold> If illegal characters appear in the input sequence, click &#x201c;predict&#x201d; bottom and a prompt page will pop up, The prompt page showing &#x201c;There is the illegal character!&#x201d;. Users can click &#x201c;submit another job.&#x201d; to return to the home page and resubmit the sequence. <bold>(C)</bold> Result display page. &#x201c;1&#x201d; in the &#x201c;Result&#x201d; column denotes that the submitted antibody candidate exhibits a high risk of hydrophobic interaction and should be excluded from the development pipeline. The &#x201c;Probability&#x201d; column represents the probability of the risk of hydrophobic interaction. The antibody will be predicted to have a high risk of hydrophobic interaction if the probability is 0.5 or higher. The result table can be sorted according to each column, and a custom display box allows users to select and display specific information as needed.</p>
</caption>
<graphic xlink:href="fgene-13-842127-g006.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>The developability assessment is performed mainly to evaluate the biochemical and biophysical properties of mAbs and to select the lead antibody with ideal efficacy, safety, pharmacokinetic characteristics, and physicochemical characteristics to meet the technical requirements of the production and preparation processes (<xref ref-type="bibr" rid="B31">Xu et&#x20;al., 2019</xref>). Various experimental strategies have been used to identify the unfavourable physicochemical properties of mAbs. However, experimental assays are time-consuming, expensive, and laborious. Computational methods can provide rapid and highly economic evaluation results and thus are expected to promote the development of antibodies (<xref ref-type="bibr" rid="B17">Krawczyk et&#x20;al., 2017</xref>). DI is a well-known in&#x20;silico tool for assessing the aggregation propensity of therapeutic antibodies and it is based on the principles that protein aggregation is mainly driven by hydrophobic interactions. Regretfully, this tool relies on the antibody structure and runs slowly. Moreover, it is an expensive tool, which makes its application limited for high-throughput screening of mAbs at the early developmental&#x20;stage.</p>
<p>Currently, data mining and machine learning are widely applied in antibody development research (<xref ref-type="bibr" rid="B6">Dzisoo et&#x20;al., 2021</xref>). <xref ref-type="bibr" rid="B19">Lecerf et&#x20;al. (2019)</xref> confirmed that the sequence characteristics of the antibody variable region can determine the physicochemical properties of therapeutic antibodies. <xref ref-type="bibr" rid="B27">Obrezanova et&#x20;al. (2015)</xref> constructed a model to predict the aggregation propensity based on the antibody sequence, and the AUC of the best AdaBoost model reached 0.76. Furthermore, <xref ref-type="bibr" rid="B12">Jain et&#x20;al. (2017a)</xref> constructed a model to predict the solvent-accessible surface area of each amino acid residue in the variable region based on the amino acid sequence of the antibody and predicted the hydrophobic interaction of antibodies through simple logistic regression. However, aforementioned tools do not provide available model or&#x20;sever.</p>
<p>The hydrophobic interaction prediction model constructed in the present study was trained on sequence only and eliminated the requirement of 3D protein structure, thereby saving the computation resources. The high sensitivity usually corresponds to the low specificity. The sensitivity of SSH2.0 reached 100.00%, which indicated that the SSH2.0 prediction result may have more false positives. However, the high sensitivity of SSH2.0 is acceptable or even preferred because the main purpose of this tool is to exclude antibodies with a risk of unfavourable hydrophobic interactions. In addition, after the step of modern mAb discovery, usually tens of thousands of therapeutic antibody candidates remain to be evaluated, and the presence of even more false positives in SSH2.0 prediction results is affordable. In summary, we propose that SSH2.0 is an efficient model for predicting the hydrophobic interaction risk of&#x20;mAbs.</p>
<p>The hydrophobic interaction risk predictor SSH2.0 constructed in this study for therapeutic mAb development is a powerful tool for selection of the antibody drug candidates with a high risk of hydrophobic interaction. This free tool based on the antibody sequence might be a better and faster alternative to the existing DI computational tool. We expect that the newer version of this tool can be used to identify reasonable mutants with a decreased risk of hydrophobic interaction. Because the number of proven therapeutic antibodies is limited, and the experiment assays vary across batches, we also expect the tool can be assessed by an independent dataset in future.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>Conclusion</title>
<p>In this study, we developed SSH2.0, a SVM-based ensemble model trained with CKSAAGP features, for predicting the hydrophobic interaction risk of therapeutic mAbs. Compared with our previous model SSH and the widely used DI tool, SSH2.0 may be a better and robust predictor that achieved the maximum sensitivity of 100.00%, and ACC and AUC of 83.97 and 88.83%, respectively. We also developed a user-friendly web server, which is freely available at <ext-link ext-link-type="uri" xlink:href="http://i.uestc.edu.cn/SSH2/">http://i.uestc.edu.cn/SSH2/</ext-link>. This tool offers a high-throughput and efficient assessment of the developability of antibodies from the perspective of hydrophobic interaction&#x20;risk.</p>
</sec>
</body>
<back>
<sec id="s6">
<title>Data Availability Statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://www.pnas.org/content/114/5/944/tab-figures-data">https://www.pnas.org/content/114/5/944/tab-figures-data</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>JH and LN conceived and designed this study. YZ and LJ wrote the manuscript. YZ, SL, and WL analyzed the data. YY wrote the interface script of web service. SX and HA drew the figures.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This work was supported by grant from the National Natural Science Foundation of China (62071099).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The handling editor declared a past collaboration with one of the authors&#x20;JH.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carter</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Lazar</surname>
<given-names>G. A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Next Generation Antibody Drugs: Pursuit of the &#x27;high-Hanging Fruit&#x27;</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>17</volume>, <fpage>197</fpage>&#x2013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1038/nrd.2017.227</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname>
<given-names>C.-C.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>C.-J.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Libsvm</article-title>. <source>ACM Trans. Intell. Syst. Technol.</source> <volume>2</volume>, <fpage>1</fpage>&#x2013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1145/1961189.1961199</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kurgan</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Prediction of Integral Membrane Protein Type by Collocated Hydrophobic Amino Acid Pairs</article-title>. <source>J.&#x20;Comput. Chem.</source> <volume>30</volume>, <fpage>163</fpage>&#x2013;<lpage>172</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.21053</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Leier</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Marquez-Lago</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>iFeature: a Python Package and Web Server for Features Extraction and Selection from Protein and Peptide Sequences</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>2499</fpage>&#x2013;<lpage>2502</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty140</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dzisoo</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Klugah-Brown</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mengesha</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>SSH: A Tool for Predicting Hydrophobic Interaction of Monoclonal Antibodies Using Sequences</article-title>. <source>Biomed. Res. Int.</source> <volume>2020</volume>, <fpage>3508107</fpage>. <pub-id pub-id-type="doi">10.1155/2020/3508107</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dzisoo</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>L. P.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>S. Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y. W.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Progress in Research on Evaluation of Developability of Therapeutic Antibody</article-title>. <source>J.&#x20;Univ. Electron. Sci. Techn. China</source> <volume>50</volume>, <fpage>476</fpage>&#x2013;<lpage>480</lpage>. </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hanke</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Klijn</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Verhaert</surname>
<given-names>P. D. E. M.</given-names>
</name>
<name>
<surname>Van Der Wielen</surname>
<given-names>L. A. M.</given-names>
</name>
<name>
<surname>Ottens</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Eppink</surname>
<given-names>M. H. M.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Prediction of Protein Retention Times in Hydrophobic Interaction Chromatography by Robust Statistical Characterization of Their Atomic-Level Surface Properties</article-title>. <source>Biotechnol. Prog.</source> <volume>32</volume>, <fpage>372</fpage>&#x2013;<lpage>381</lpage>. <pub-id pub-id-type="doi">10.1002/btpr.2219</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ru</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>SABinder: A Web Service for Predicting Streptavidin-Binding Peptides</article-title>. <source>Biomed. Res. Int.</source> <volume>2016</volume>, <fpage>9175143</fpage>. <pub-id pub-id-type="doi">10.1155/2016/9175143</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>PhD7Faster 2.0: Predicting Clones Propagating Faster from the Ph.D.-7 Phage Display Library by Coupling PseAAC and Tripeptide Composition</article-title>. <source>PeerJ</source> <volume>7</volume>, <fpage>e7131</fpage>. <pub-id pub-id-type="doi">10.7717/peerj.7131</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>MRMD2.0: A Python Tool for Machine Learning with Feature Ranking and Reduction</article-title>. <source>Curr. Bioinform.</source> <volume>15</volume>, <fpage>1213</fpage>&#x2013;<lpage>1221</lpage>. <pub-id pub-id-type="doi">10.2174/1574893615999200503030350</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hebditch</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Roche</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Curtis</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Warwicker</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Models for Antibody Behavior in Hydrophobic Interaction Chromatography and in Self-Association</article-title>. <source>J.&#x20;Pharm. Sci.</source> <volume>108</volume>, <fpage>1434</fpage>&#x2013;<lpage>1441</lpage>. <pub-id pub-id-type="doi">10.1016/j.xphs.2018.11.035</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jain</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Boland</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lilov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Burnina</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2017a</year>). <article-title>Prediction of Delayed Retention of Antibodies in Hydrophobic Interaction Chromatography from Sequence Using Machine Learning</article-title>. <source>Bioinformatics</source> <volume>33</volume>, <fpage>3758</fpage>&#x2013;<lpage>3766</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx519</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jain</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Durand</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Houston</surname>
<given-names>N. R.</given-names>
</name>
<name>
<surname>Nett</surname>
<given-names>J.&#x20;H.</given-names>
</name>
<etal/>
</person-group> (<year>2017b</year>). <article-title>Biophysical Properties of the Clinical-Stage Antibody Landscape</article-title>. <source>Proc. Natl. Acad. Sci. USA</source> <volume>114</volume>, <fpage>944</fpage>&#x2013;<lpage>949</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1616408114</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>NeuroPP: A Tool for the Prediction of Neuropeptide Precursors Based on Optimal Sequence Composition</article-title>. <source>Interdiscip. Sci.</source> <volume>11</volume>, <fpage>108</fpage>&#x2013;<lpage>114</lpage>. <pub-id pub-id-type="doi">10.1007/s12539-018-0287-2</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kapingidza</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Kowal</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chruszcz</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Antigen-Antibody Complexes</article-title>. <source>Subcell Biochem.</source> <volume>94</volume>, <fpage>465</fpage>&#x2013;<lpage>497</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-41769-7_19</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaplon</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Muralidharan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Reichert</surname>
<given-names>J.&#x20;M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Antibodies to Watch in 2020</article-title>. <source>MAbs</source> <volume>12</volume>, <fpage>1703531</fpage>. <pub-id pub-id-type="doi">10.1080/19420862.2019.1703531</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krawczyk</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Dunbar</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Deane</surname>
<given-names>C. M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Computational Tools for Aiding Rational Antibody Design</article-title>. <source>Methods Mol. Biol.</source> <volume>1529</volume>, <fpage>399</fpage>&#x2013;<lpage>416</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-4939-6637-0_21</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lauer</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Chennamsetty</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Egodage</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Helk</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Trout</surname>
<given-names>B. L.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Developability index: a Rapid In Silico Tool for the Screening of Antibody Aggregation Propensity</article-title>. <source>J.&#x20;Pharm. Sci.</source> <volume>101</volume>, <fpage>102</fpage>&#x2013;<lpage>115</lpage>. <pub-id pub-id-type="doi">10.1002/jps.22758</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lecerf</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kanyavuz</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lacroix-Desmazes</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dimitrov</surname>
<given-names>J.&#x20;D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Sequence Features of Variable Region Determining Physicochemical Properties and Polyreactivity of Therapeutic Antibodies</article-title>. <source>Mol. Immunol.</source> <volume>112</volume>, <fpage>338</fpage>&#x2013;<lpage>346</lpage>. <pub-id pub-id-type="doi">10.1016/j.molimm.2019.06.012</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Z. X.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>AbRSA: A Robust Tool for Antibody Numbering</article-title>. <source>Protein Sci.</source> <volume>28</volume>, <fpage>1524</fpage>&#x2013;<lpage>1531</lpage>. <pub-id pub-id-type="doi">10.1002/pro.3633</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>PSBinder: A Web Service for Predicting Polystyrene Surface-Binding Peptides</article-title>. <source>Biomed. Res. Int.</source> <volume>2017</volume>, <fpage>5761517</fpage>. <pub-id pub-id-type="doi">10.1155/2017/5761517</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lienqueo</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Mahn</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Navarro</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Salgado</surname>
<given-names>J.&#x20;C.</given-names>
</name>
<name>
<surname>Perez-Acle</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Rapaport</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>New Approaches for Predicting Protein Retention Time in Hydrophobic Interaction Chromatography</article-title>. <source>J.&#x20;Mol. Recognit.</source> <volume>19</volume>, <fpage>260</fpage>&#x2013;<lpage>269</lpage>. <pub-id pub-id-type="doi">10.1002/jmr.776</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>R.-M.</given-names>
</name>
<name>
<surname>Hwang</surname>
<given-names>Y.-C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>I.-J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>C.-C.</given-names>
</name>
<name>
<surname>Tsai</surname>
<given-names>H.-Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.-J.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Development of Therapeutic Antibodies for the Treatment of Diseases</article-title>. <source>J.&#x20;Biomed. Sci.</source> <volume>27</volume>, <fpage>1</fpage>. <pub-id pub-id-type="doi">10.1186/s12929-019-0592-z</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahn</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lienqueo</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Salgado</surname>
<given-names>J.&#x20;C.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Methods of Calculating Protein Hydrophobicity and Their Application in Developing Correlations to Predict Hydrophobic Interaction Chromatography Retention</article-title>. <source>J.&#x20;Chromatogr. A</source> <volume>1216</volume>, <fpage>1838</fpage>&#x2013;<lpage>1844</lpage>. <pub-id pub-id-type="doi">10.1016/j.chroma.2008.11.089</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Martinez Morales</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zalar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sonzini</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Golovanov</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Van Der Walle</surname>
<given-names>C. F.</given-names>
</name>
<name>
<surname>Derrick</surname>
<given-names>J.&#x20;P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Interaction of a Macrocycle with an Aggregation-Prone Region of a Monoclonal Antibody</article-title>. <source>Mol. Pharm.</source> <volume>16</volume>, <fpage>3100</fpage>&#x2013;<lpage>3108</lpage>. <pub-id pub-id-type="doi">10.1021/acs.molpharmaceut.9b00338</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ning</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Abagna</surname>
<given-names>H. B.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Development and Application of Therapeutic Antibodies against COVID-19</article-title>. <source>Int. J.&#x20;Biol. Sci.</source> <volume>17</volume>, <fpage>1486</fpage>&#x2013;<lpage>1496</lpage>. <pub-id pub-id-type="doi">10.7150/ijbs.59149</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Obrezanova</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Arnell</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>De La Cuesta</surname>
<given-names>R. G.</given-names>
</name>
<name>
<surname>Berthelot</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Gallagher</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Zurdo</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Aggregation Risk Prediction for Antibodies and its Application to Biotherapeutic Development</article-title>. <source>MAbs</source> <volume>7</volume>, <fpage>352</fpage>&#x2013;<lpage>363</lpage>. <pub-id pub-id-type="doi">10.1080/19420862.2015.1007828</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patel</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Shukla</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ussery</surname>
<given-names>D. W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Machine Learning Methods in Drug Discovery</article-title>. <source>Molecules</source> <volume>25</volume>. <pub-id pub-id-type="doi">10.3390/molecules25225277</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Romero-Molina</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ruiz-Blanco</surname>
<given-names>Y. B.</given-names>
</name>
<name>
<surname>Harms</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>M&#xfc;nch</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sanchez-Garcia</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>PPI-detect: A Support Vector Machine Model for Sequence-Based Prediction of Protein-Protein Interactions</article-title>. <source>J.&#x20;Comput. Chem.</source> <volume>40</volume>, <fpage>1233</fpage>&#x2013;<lpage>1242</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.25780</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>NeuroCS: A Tool to Predict Cleavage Sites of Neuropeptide Precursors</article-title>. <source>Protein Pept. Lett.</source> <volume>27</volume>, <fpage>337</fpage>&#x2013;<lpage>345</lpage>. <pub-id pub-id-type="doi">10.2174/0929866526666191112150636</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mason</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Rossomando</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Structure, Heterogeneity and Developability Assessment of Therapeutic Antibodies</article-title>. <source>MAbs</source> <volume>11</volume>, <fpage>239</fpage>&#x2013;<lpage>264</lpage>. <pub-id pub-id-type="doi">10.1080/19420862.2018.1553476</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Colorectal Cancer Diagnostic Algorithm Based on Sub-patch Weight Color Histogram in Combination of Improved Least Squares Support Vector Machine for Pathological Image</article-title>. <source>J.&#x20;Med. Syst.</source> <volume>43</volume>, <fpage>306</fpage>. <pub-id pub-id-type="doi">10.1007/s10916-019-1429-8</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>