<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Chem.</journal-id>
<journal-title>Frontiers in Chemistry</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Chem.</abbrev-journal-title>
<issn pub-type="epub">2296-2646</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">753002</article-id>
<article-id pub-id-type="doi">10.3389/fchem.2021.753002</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Chemistry</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>OnionNet-2: A Convolutional Neural Network Model for Predicting Protein-Ligand Binding Affinity Based on Residue-Atom Contacting Shells</article-title>
<alt-title alt-title-type="left-running-head">Wang et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">OnionNet-2 PL Binding Affinity Prediction</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Zechen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1430505/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zheng</surname>
<given-names>Liangzhen</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Yang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Qu</surname>
<given-names>Yuanyuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Yong-Qiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Mingwen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Mu&#x2009;</surname>
<given-names>Yuguang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/59932/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li&#x2009;</surname>
<given-names>Weifeng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/378854/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<label>
<sup>1</sup>
</label>School of Physics, Shandong University, <addr-line>Jinan</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<label>
<sup>2</sup>
</label>Tencent AI Lab, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<label>
<sup>3</sup>
</label>School of Biological Sciences, Nanyang Technological University, <country>Singapore</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/265183/overview">Yong Wang</ext-link>, Ningbo University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/754964/overview">Martin Zacharias</ext-link>, Technical University of Munich, Germany</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1480119/overview">Jian Zhang</ext-link>, Nanjing University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yuguang Mu&#x2009;, <email>ygmu@ntu.edu.sg</email>; Weifeng Li&#x2009;, <email>lwf@sdu.edu.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Theoretical and Computational Chemistry, a section of the journal Frontiers in Chemistry</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>27</day>
<month>10</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>9</volume>
<elocation-id>753002</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>08</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>06</day>
<month>10</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2021 Wang, Zheng, Liu, Qu, Li, Zhao, Mu&#x2009; and Li&#x2009;.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Wang, Zheng, Liu, Qu, Li, Zhao, Mu&#x2009; and Li&#x2009;</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>One key task in virtual screening is to accurately predict the binding affinity (&#x25b3;<italic>G</italic>) of protein-ligand complexes. Recently, deep learning (DL) has significantly increased the predicting accuracy of scoring functions due to the extraordinary ability of DL to extract useful features from raw data. Nevertheless, more efforts still need to be paid in many aspects, for the aim of increasing prediction accuracy and decreasing computational cost. In this study, we proposed a simple scoring function (called OnionNet-2) based on convolutional neural network to predict &#x25b3;<italic>G</italic>. The protein-ligand interactions are characterized by the number of contacts between protein residues and ligand atoms in multiple distance shells. Compared to published models, the efficacy of OnionNet-2 is demonstrated to be the best for two widely used datasets CASF-2016 and CASF-2013 benchmarks. The OnionNet-2 model was further verified by non-experimental decoy structures from docking program and the CSAR NRC-HiQ data set (a high-quality data set provided by CSAR), which showed great success. Thus, our study provides a simple but efficient scoring function for predicting protein-ligand binding free energy.</p>
</abstract>
<kwd-group>
<kwd>protein-ligand binding</kwd>
<kwd>deep learning</kwd>
<kwd>onionnet</kwd>
<kwd>residue-atom distance</kwd>
<kwd>structure-based affinity prediction</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Protein-ligand binding is the basic of almost all processes in living organisms <xref ref-type="bibr" rid="B8">Du et&#x20;al. (2016)</xref> thus predicting binding affinity (&#x25b3;<italic>G</italic>) of protein-ligand complex becomes the research focus of bioinformatics and drug design <xref ref-type="bibr" rid="B19">Guedes et&#x20;al. (2018)</xref>; <xref ref-type="bibr" rid="B20">Guvench and MacKerell (2009)</xref>; <xref ref-type="bibr" rid="B11">Ellingson et&#x20;al. (2020)</xref>. Theoretically, molecular dynamics (MD) simulations and free energy calculations (for instance, thermal integration method and free energy perturbation can provide accurate predictions of &#x25b3;<italic>G</italic> relying on extensive configurational sampling and calculation, leading to a large demand in computational cost <xref ref-type="bibr" rid="B11">Ellingson et&#x20;al. (2020)</xref>; <xref ref-type="bibr" rid="B36">Michel and Essex (2010)</xref>; <xref ref-type="bibr" rid="B16">Gilson and Zhou (2007)</xref>; <xref ref-type="bibr" rid="B21">Hansen and Van Gunsteren (2014)</xref>. Therefore, developing simple, accurate and efficient scoring methods to predict protein-ligand binding will greatly accelerate the drug design process <xref ref-type="bibr" rid="B33">Liu and Wang (2015)</xref>. To achieve this, several theoretical methods (scoring functions) have been proposed. Typically, the scoring functions are based on calculations of interactions between protein and ligand atoms <xref ref-type="bibr" rid="B8">Du et&#x20;al. (2016)</xref>; <xref ref-type="bibr" rid="B33">Liu and Wang (2015)</xref>; <xref ref-type="bibr" rid="B24">Huang et&#x20;al. (2010)</xref>; <xref ref-type="bibr" rid="B18">Grinter and Zou (2014)</xref>. This includes quantum mechanics calculations, molecular dynamics simulations (electrostatic interaction, van der Waals interaction, hydrogen-bond and etc.), empirical-based interacting models. <xref ref-type="bibr" rid="B8">Du et&#x20;al. (2016)</xref>; <xref ref-type="bibr" rid="B24">Huang et&#x20;al. (2010)</xref>; <xref ref-type="bibr" rid="B18">Grinter and Zou (2014)</xref>; <xref ref-type="bibr" rid="B23">Huang et&#x20;al. (2006)</xref>.</p>
<p>In recent years, approaches based on machine learning (ML) have been applied in scoring functions and demonstrated great success <xref ref-type="bibr" rid="B34">Lo et&#x20;al. (2018)</xref>; <xref ref-type="bibr" rid="B54">Vamathevan et&#x20;al. (2019)</xref>; <xref ref-type="bibr" rid="B44">Shen et&#x20;al. (2020a)</xref>; <xref ref-type="bibr" rid="B29">Lavecchia (2015)</xref>. For instance, RF-Score <xref ref-type="bibr" rid="B5">Ballester and Mitchell (2010)</xref> and NNScore are two pioneering ML-based scoring functions <xref ref-type="bibr" rid="B10">Durrant and McCammon (2010)</xref>. Compared with classical approaches, these ML-based methods allow higher flexibility in selecting configurational representations or features for protein and ligand. More importantly, these methods have been demonstrated to perform better and more effective than classical approaches <xref ref-type="bibr" rid="B2">Ain et&#x20;al. (2015)</xref>; S <xref ref-type="bibr" rid="B22">Heck et&#x20;al. (2017)</xref>. Recently, the deep learning (DL) approaches have provided alternative solution. Compared with ML, the DL models perform better at learning features from the raw data to extract the relationship between these features and labels. <xref ref-type="bibr" rid="B56">Wang and Gao (2019)</xref>; <xref ref-type="bibr" rid="B14">Gawehn et&#x20;al. (2016)</xref>; <xref ref-type="bibr" rid="B7">Chen et&#x20;al. (2018)</xref> Thus, DL algorithms have been introduced to model the structure-activity relationships <xref ref-type="bibr" rid="B58">Yang et&#x20;al. (2019)</xref>; <xref ref-type="bibr" rid="B57">Winter et&#x20;al. (2019)</xref>; <xref ref-type="bibr" rid="B15">Ghasemi et&#x20;al. (2018)</xref>. One of the most popular methods of DL is the convolutional neural network (CNN), which is a multi-layer perceptron inspired by the neural network of living organisms <xref ref-type="bibr" rid="B28">Lavecchia (2019)</xref>.</p>
<p>Inspired by the great success of DL and CNN techniques, several models applying CNN to predict protein-ligand interaction have been reported <xref ref-type="bibr" rid="B26">Jim&#xe9;nez et&#x20;al. (2017)</xref>; <xref ref-type="bibr" rid="B17">Gomes et&#x20;al. (2017)</xref>; <xref ref-type="bibr" rid="B41">&#xd6;zt&#xfc;rk et&#x20;al. (2018)</xref>; <xref ref-type="bibr" rid="B49">Stepniewska-Dziubinska et&#x20;al. (2018)</xref>; <xref ref-type="bibr" rid="B27">Jim&#xe9;nez et&#x20;al. (2018)</xref>; <xref ref-type="bibr" rid="B51">Torng and Altman (2019a)</xref>; <xref ref-type="bibr" rid="B60">Zheng et&#x20;al. (2019)</xref>; <xref ref-type="bibr" rid="B38">Morrone et&#x20;al. (2020)</xref>; <xref ref-type="bibr" rid="B52">Torng and Altman (2019b)</xref>. For example, &#xd6;zt&#xfc;rk and co-workers reported a DeepDTA model based on one-dimensional (1D) convolution, which took protein sequences and simplified molecular input line entry specification (SMILES) codes of ligand as inputs to predict drug-target &#x25b3;<italic>G</italic> <xref ref-type="bibr" rid="B41">&#xd6;zt&#xfc;rk et&#x20;al. (2018)</xref>. Using 3D CNN model, two independent groups developed scoring functions, named Pafnucy <xref ref-type="bibr" rid="B49">Stepniewska-Dziubinska et&#x20;al. (2018)</xref> and <italic>K</italic>
<sub>
<italic>deep</italic>
</sub> <xref ref-type="bibr" rid="B27">Jim&#xe9;nez et&#x20;al. (2018)</xref>, to model the complex in a cubic box centered on the geometric center of the ligand to predict the &#x25b3;<italic>G</italic> of protein-ligand complex. More interestingly, Russ et&#x20;al. employed Graph-CNNs to automatically extract features from protein pocket and 2D ligand graphs, and demonstrated that the Graph-CNN framework can achieve superior performance without relying on protein-ligand complexes <xref ref-type="bibr" rid="B51">Torng and Altman (2019a)</xref>. Our group has proposed a 2D convolution-based predictor, called OnionNet, based on element-pair-specific contacts between ligands and protein atoms <xref ref-type="bibr" rid="B60">Zheng et&#x20;al. (2019)</xref>. As is shown, these DL and CNN based approaches, achieved higher accuracy in &#x25b3;<italic>G</italic> prediction than most traditional scoring functions, such as Auto Dock <xref ref-type="bibr" rid="B37">Morris et&#x20;al. (1998)</xref>; <xref ref-type="bibr" rid="B25">Huey et&#x20;al. (2007)</xref>, X-Score. <xref ref-type="bibr" rid="B55">Wang et&#x20;al. (2002)</xref> and KScore. <xref ref-type="bibr" rid="B59">Zhao et&#x20;al. (2008)</xref>.</p>
<p>Physically, the dominating factors for overall binding affinity involve electrostatic interactions, van der Waals interactions, hydrogen bonds, hydration/de-hydration during complexation. Instead, for DL scoring functions, how to treat with the high-dimensional structural information encoded in the 3D structures and convert to the low-dimensional features for ML (or DL) training is critical. For most structure-based ML/DL models, the features are usually derived from the atomic information of proteins and ligands, such as the element type and spatial coordinates of the atom and even other atomic properties <xref ref-type="bibr" rid="B49">Stepniewska-Dziubinska et&#x20;al. (2018)</xref>; <xref ref-type="bibr" rid="B27">Jim&#xe9;nez et&#x20;al. (2018)</xref>. In our OnionNet model, we characterized the protein-ligand interactions by the number of element-pair-specific contacts in multiple distance shells. <xref ref-type="bibr" rid="B60">Zheng et&#x20;al. (2019)</xref>; <xref ref-type="bibr" rid="B46">Song et&#x20;al. (2020)</xref>.</p>
<p>As we all know, the same elements in different residues have quite different physical and chemical properties, which might greatly affect the protein-ligand binding event. Therefore, it may be insufficient enough to characterize the intrinsic physical and chemical properties of proteins by dividing the protein into eight types of atoms. Considering that the twenty types of amino acids can be treated as intrinsic classification of protein compounds which involve lay features of them, like polar, apolar, aromatics, and etc. It may be more reasonable to characterize the physicochemical properties of proteins through individual residues. Especially the residues in the binding pocket, they directly participate in the construction of the binding environment, which plays a decisive role in ligand binding events. It is undeniable that using residues as the basic unit actually &#x201c;coarse-grained&#x201d; the protein, which will lose part of the structural information, but this may also help to improve the generalization ability of the model itself. In view of these, we anticipate that it may be more beneficial to encode protein as residues instead of atoms in developing DL scoring functions.</p>
<p>In this work, we proposed a simple OnionNet-2&#x2014;a 2D CNN based regression model to predict protein-ligand &#x25b3;<italic>G</italic>, which adopts the rotation-free residue-atom-specific contacts in multiple distance shells to describe the protein (residues) - ligand (atoms) interactions. The model was trained on the PDBbind database <xref ref-type="bibr" rid="B30">Li et&#x20;al. (2014a)</xref> and tested by the comparative assessment of scoring function (CASF) benchmarks, where CASF-2016 <xref ref-type="bibr" rid="B50">Su et&#x20;al. (2018)</xref> is employed as the primary benchmark. When the total number of shells was 62, OnionNet-2 achieved the best performance with the Pearson correlation coefficient (<italic>R</italic>) reaching 0.864 and a root-mean-squared error (RMSE) of 1.164. When applied to the earlier version of CASF-2013, <xref ref-type="bibr" rid="B30">Li et&#x20;al. (2014a)</xref>, <xref ref-type="bibr" rid="B31">Li et&#x20;al. (2014b)</xref>, <xref ref-type="bibr" rid="B32">Li et&#x20;al. (2018)</xref> our present model achieved R of 0.821 and RMSE of 1.357. An additional high-quality data set, the CSAR NRC-HiQ data set (consisting of two subsets) <xref ref-type="bibr" rid="B9">Dunbar et&#x20;al. (2011)</xref> was also used to verify OnionNet-2. Our model achieved <italic>R</italic> of 0.89 for NRC-HiQ subset 1 (55 complexes) and 0.87 for NRC-HiQ subset 2 (49 complexes). The performance is indeed higher than two previous ML scoring models, RF-Score (<italic>R</italic>&#x20;&#x3d; 0.78 and 0.75, respectively) and <italic>K</italic>
<sub>
<italic>deep</italic>
</sub> (<italic>R</italic>&#x20;&#x3d; 0.72 and 0.65, respectively). We demonstrated that, our present method can significantly improve the prediction power by about 3.7% than previous models, thus providing an efficient and accurate approach for predicting protein-ligand interactions and uncover a new trend of using DL technique for massive biological structures training for drug design.</p>
</sec>
<sec id="s2">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 Evaluation Metrics</title>
<p>We used Pearson correlation coefficient R, RMSE and Standard Deviation (SD) to evaluate the scoring power of the model which are defined as:<disp-formula id="e1">
<mml:math id="m1">
<mml:mi>R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msqrt>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(1)</label>
</disp-formula>
<disp-formula id="e2">
<mml:math id="m2">
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m3">
<mml:mi mathvariant="italic">SD</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msqrt>
</mml:math>
<label>(3)</label>
</disp-formula>where x<sub>
<italic>i</italic>
</sub> is the predicted pK<sub>
<italic>d</italic>
</sub> for <italic>i</italic>th complex; y<sub>
<italic>i</italic>
</sub> is the experimental pK<sub>
<italic>d</italic>
</sub> of this complex; <inline-formula id="inf1">
<mml:math id="m4">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf2">
<mml:math id="m5">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> are the averages of all predicted values and experimental values; a and b are the intercept and the slope of the regression line, respectively <xref ref-type="bibr" rid="B50">Su et&#x20;al. (2018)</xref>.</p>
</sec>
<sec id="s2-2">
<title>2.2 Preparation of Dataset</title>
<p>We mainly used the protein-ligand complexes of PDBbind database v.2019 (<ext-link ext-link-type="uri" xlink:href="http://www.pdbbind-cn.org/">http://www.pdbbind-cn.org/</ext-link>) for training. This database consists of two overlapping subsets, the general set and the refined set. The general set includes all available complexes and the refined set comprises protein-ligand complexes with high-quality structure and binding information selected from the general set. For each structure of the protein-ligand complex, the corresponding binding affinity is represented by the negative logarithms (pK<sub>
<italic>d</italic>
</sub>) of the dissociation constants (<italic>Kd</italic>), inhibition constants (<italic>Ki</italic>) or half inhibition concentrations (<italic>IC</italic>50). In order to evaluate the predictive ability and compare with other scoring functions, OnionNet-2 was evaluated on the CASF-2016 test set (core set v.2016) <xref ref-type="bibr" rid="B50">Su et&#x20;al. (2018)</xref> and CASF-2013 test set (core set v.2013) <xref ref-type="bibr" rid="B30">Li et&#x20;al. (2014a)</xref>, <xref ref-type="bibr" rid="B31">Li et&#x20;al. (2014b)</xref>, <xref ref-type="bibr" rid="B32">Li et&#x20;al. (2018)</xref>. It should be noted that the CASF-2016 test set is the latest update of CASF-2016, which contains 285&#x20;high-quality complexes. While for core set v.2013, it is a subset of the PDBbind database v.2013, consisting of 195&#x20;protein-ligand complexes classified in 65 clusters with binding constants spanning nearly 10 orders of magnitude. Besides, a data set called CSAR NRC-HiQ, consisting of two subsets containing 176 and 167 complexes respectively, <xref ref-type="bibr" rid="B9">Dunbar et&#x20;al. (2011)</xref> was employed as a third test set. For the previous models of Kdeep and RF-score, 55 and 49 complexes in two subsets were used as test data <xref ref-type="bibr" rid="B27">Jim&#xe9;nez et&#x20;al. (2018)</xref>. To provide a direct comparison with <italic>K</italic>
<sub>
<italic>deep</italic>
</sub> and RF-score, we adopted the same data for the OnionNet-2&#x20;test.</p>
<p>In order to perform normal training and testing, it is necessary to redistribute remaining complexes in PDBbind database v.2019. First, we excluded the complexes contained in three test sets from PDBbind database v.2019 (general set and refined set). Then, as a common practice (Reference: Pafnucy <xref ref-type="bibr" rid="B49">Stepniewska-Dziubinska et&#x20;al. (2018)</xref> and OnionNet <xref ref-type="bibr" rid="B60">Zheng et&#x20;al. (2019)</xref>), 1,000 complexes were randomly sampled from v.2019 refined set (after filtering all complexes used in the test sets described previously) as the validating set. Finally, the remaining complexes (that is, the complexes that are not included in the three test sets and validating set) were adopted for the training set. This ensures that there is no overlapping protein-ligand complex in the training set, validating set and test&#x20;sets.</p>
</sec>
<sec id="s2-3">
<title>2.3 Descriptors</title>
<p>The features employed are the pair numbers of the specific residue (protein)-atom (ligand) combination in multiple distance shells. The minimum distances between any atom in the ligand and any residue of protein are treated as the representative distances. First, around each atom in the ligand, we defined N continuously packed shells. The thickness of each shell is <italic>&#x3b4;</italic>, except that the first shell is a sphere with a radius of <italic>d</italic>
<sub>0</sub>. The boundary K<sub>
<italic>i</italic>
</sub> of the <italic>i</italic>th shell is as follows<disp-formula id="equ1">
<mml:math id="m6">
<mml:mtable class="align-star" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mn>0</mml:mn>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="0.17em"/>
<mml:mspace width="0.3333em"/>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mspace width="0.3333em"/>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:mspace width="0.3333em"/>
<mml:mi>i</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
</p>
<p>Meanwhile, we classified atoms in the ligand into eight types, namely C, H, O, N, P, S, HAL, and DU, where HAL represents the halogen elements (F, Cl, Br, and I), and DU represents the element types excluded in these seven types.<disp-formula id="equ2">
<mml:math id="m7">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.3333em"/>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mi mathvariant="normal">U</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>When pre-processing the structure file, water and ions were treated explicitly because crystal water molecules and ions could affect the protein-ligand binding <xref ref-type="bibr" rid="B13">Garc&#xed;a-Sosa (2013)</xref>; <xref ref-type="bibr" rid="B47">Spyrakis et&#x20;al. (2017)</xref>. In addition to the twenty standard residues, we added an expanded type named &#x201c;OTH&#x201d; to represent water, ions and any other non-standard residues.<disp-formula id="equ3">
<mml:math id="m8">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.3333em"/>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">Y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">V</mml:mi>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mi mathvariant="normal">U</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">I</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">Y</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">Y</mml:mi>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">U</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">Y</mml:mi>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mi mathvariant="normal">I</mml:mi>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mi mathvariant="normal">H</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>It is worth mention that the residue-atom distance is defined as the distance between the atom in the ligand and the nearest heavy atom in the residue. A 2D visual representation is depicted at the upper left of <xref ref-type="fig" rid="F1">Figure&#x20;1</xref>. For any shell, the number of contacts for each residue-atom pair is calculated and used as a feature. Each shell has 8&#x20;&#xd7; 21&#x20;&#x3d; 168&#x20;residue-atom combinations, which means that there are 168 features for a shell. Thus, if the total number of shells is <italic>N</italic>, 168&#x20;&#xd7; <italic>N</italic> features will be generated.<disp-formula id="e4">
<mml:math id="m9">
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msubsup>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m10">
<mml:msubsup>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mtable class="gathered">
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Workflow of OnionNet-2. The features based on the residue-atom contacts are converted into a 2D image and fed to the CNN architecture.</p>
</caption>
<graphic xlink:href="fchem-09-753002-g001.tif"/>
</fig>
<p>Here, <italic>R</italic> is the total number of residues in the protein, and <italic>E</italic> is the total number of atoms in the ligand. The <italic>d</italic>
<sub>
<italic>r</italic>,<italic>e</italic>
</sub> is the minimum distance between the residue <italic>r</italic> in the protein and the atom <italic>e</italic> in the ligand, and <inline-formula id="inf3">
<mml:math id="m11">
<mml:msubsup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is the number of contacts of the specific residue-element combination in the <italic>i</italic>th shell. The <inline-formula id="inf4">
<mml:math id="m12">
<mml:msubsup>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is 1 when (<italic>i</italic>&#x20;&#x2212; 2)<italic>&#x3b4;</italic> &#x2b; <italic>d</italic>
<sub>0</sub> &#x2264; <italic>d</italic>
<sub>
<italic>r</italic>,<italic>e</italic>
</sub> &#x3c; (<italic>i</italic>&#x20;&#x2212; 1)<italic>&#x3b4;</italic> &#x2b; <italic>d</italic>
<sub>0</sub>, otherwise <inline-formula id="inf5">
<mml:math id="m13">
<mml:msubsup>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is 0. Following our previous study, <xref ref-type="bibr" rid="B60">Zheng et&#x20;al. (2019)</xref> we used <italic>d</italic>
<sub>0</sub> &#x3d; 1&#xa0;&#xc5; and <italic>&#x3b4;</italic> &#x3d; 0.5&#xa0;&#xc5;. Interestingly such shell-like, or radial, representations of protein environments, have been demonstrated to be superior features in protein function prediction <xref ref-type="bibr" rid="B52">Torng and Altman (2019b)</xref>. The source code of OnionNet-2 is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/zchwang/OnionNet-2/">https://github.com/zchwang/OnionNet-2/</ext-link>.</p>
</sec>
<sec id="s2-4">
<title>2.4 Architecture</title>
<p>We adopted a CNN model based on 2D convolution to learn the relationship between the contact features and the &#x25b3;<italic>G</italic>. The model was constructed using the Keras package in tensorflow <xref ref-type="bibr" rid="B1">Abadi et&#x20;al. (2016)</xref>. The workflow architecture is shown in <xref ref-type="fig" rid="F1">Figure&#x20;1</xref>.</p>
<p>The raw data is pre-processed before input into the CNN model. Here, the features are standardized through the scikit-learn package <xref ref-type="bibr" rid="B43">Pedregosa et&#x20;al. (2011)</xref>, and the processed features confirmed the standard normal distribution. Considering the big success of CNN model using 2D convolution in image recognition, <xref ref-type="bibr" rid="B42">Pak and Kim (2017)</xref> the 1D vector containing the protein-ligand interactions was converted into a 2D matrix to mimic the feature images, which was used as the input of CNN&#x20;model.</p>
<p>CNNs usually consist of multiple layers with different functions, and the convolution layer is the key part of CNN models, which is to extract different features from input data <xref ref-type="bibr" rid="B28">Lavecchia (2019)</xref>. The filter, also called the convolution kernel, is the core part of the convolutional layer, and the local features of the input &#x201c;picture&#x201d; are extracted through the sliding of the filter <xref ref-type="bibr" rid="B14">Gawehn et&#x20;al. (2016)</xref>; <xref ref-type="bibr" rid="B4">Angermueller et&#x20;al. (2016)</xref>. In the OnionNet-2 model, we used three convolutional layers, with 32, 64, and 128 filters respectively and the filter sizes were all set as 4, with strides as 1. The results of the last convolutional layer need to be flattened before being passed to the fully connected layers. The fully connected layers can integrate the local features extracted by the convolutional layers to give the prediction of pK<sub>
<italic>d</italic>
</sub> value. Increasing the width and length of the fully connected layer can improve the complexity and nonlinear expression ability of the model, but in practice, this may lead to unexpected overfitting. In addition, increasing parameters will significantly increase the computational cost. After preliminary tests, two fully connected layers with 100 and 50 neurons are used before the output layer, which is capable of capturing the nonlinear relationship between the features and the pK<sub>
<italic>d</italic>
</sub> values.</p>
<p>To further increase the nonlinear ability of the model, a rectified linear unit (RELU) layer was added after each convolutional layer and fully connected layer. Also, a batch normalization layer was used after the fully connected layer. The stochastic gradient descent (SGD) optimizer was adopted and the learning rate was set as 0.001. To reduce overfitting, L2 regularization with weight decay <italic>&#x3bb;</italic>&#x20;&#x3d; 0.01 was used after each fully connected layer. The number of samples processed per batch is&#x20;64.</p>
<p>To evaluate the performance of the OnionNet-2, we adopted the loss function defined in the previous work <xref ref-type="bibr" rid="B60">Zheng et&#x20;al. (2019)</xref>.<disp-formula id="e6">
<mml:math id="m14">
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:math>
<label>(6)</label>
</disp-formula>where R and RMSE represent Pearson correlation coefficient (<xref ref-type="disp-formula" rid="e1">Eq.&#x20;1</xref>) and root-mean-squared error (<xref ref-type="disp-formula" rid="e2">Eq. 2</xref>), respectively. The <italic>&#x3b1;</italic>(0 &#x2264; <italic>&#x3b1;</italic> &#x2264; 1) value is an adjustable factor for adjusting the weight with <italic>R</italic> and RMSE, which was finally set to 0.7. For each independent training task, we adopted early stopping (patience &#x3d; 20, that is, if the change of the loss value in the validating set is less than 0.001 after 20 epochs, the training is terminated) and save the model that performed best on the validating set. For the prediction in each case, five independently trainings were conducted to obtain the predicted mean&#x20;value.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<title>3 Results and Discussions</title>
<sec id="s3-1">
<title>3.1 The Predictive Power of OnionNet-2</title>
<p>Firstly, we explored the effect of shell number <italic>N</italic> on the predictive capability of the OnionNet-2 model. A range of the total shell number 10, &#x2264;, <italic>N</italic>&#x20;&#x2264; 90 was tested with interval of 2. According to our definitions of distance shell, this covers a separation between the residue and the atom from 0.55 to 4.55&#xa0;nm. <xref ref-type="fig" rid="F2">Figure&#x20;2</xref> depicts the trend of the <italic>R</italic> value to shell number <italic>N</italic> testing with core set v.2016. For <italic>N</italic> from 10 to 20, the <italic>R</italic> quickly increases as the total number of shells increases. This is expected because as the number of shells increases, the interactions between ligand and protein were gradually captured by the model. The <italic>R</italic> value reached the first peak for <italic>N</italic> is 30. This means that OnionNet-2 can achieve high prediction accuracy at a relatively low computational cost. Then, <italic>R</italic> fluctuates in a range of 0.01 until reaches the global maximum value when <italic>N</italic>&#x20;&#x3d; 62. <xref ref-type="fig" rid="F3">Figure&#x20;3</xref> summarized the mean predicted value of each complex from five independent training, with respect to experimental value, using <italic>N</italic>&#x20;&#x3d; 62, on the training set, validating set and two testing sets, core set v.2016 and core set v.2013. It shows that the predicted pK<sub>
<italic>d</italic>
</sub> and experimental pK<sub>
<italic>d</italic>
</sub> are highly correlated for the two testing sets and validating set. After this point, <italic>R</italic> decreases when <italic>N</italic> increase. We attribute this to the enormous data that leads to the introduction of noise in the training. Unless otherwise specified, we adopted <italic>N</italic> of 62 in the following discussions. In addition, we also re-trained the model with two elder versions (v.2016 and v.2018) of the PDBbind database, and the <italic>R</italic> values of our re-trained models are almost the same (<xref ref-type="sec" rid="s10">Supplementary Figure S1</xref> and <xref ref-type="sec" rid="s10">Supplementary Table&#x20;S1</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The Pearson correlation coefficient with respect to the shell number <italic>N</italic> for OnionNet-2 testing with core set v.2016. The bars indicate the standard deviations of the <italic>R</italic> values for five independent&#x20;runs.</p>
</caption>
<graphic xlink:href="fchem-09-753002-g002.tif"/>
</fig>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Results of OnionNet-2 model (<italic>N</italic>&#x20;&#x3d; 62). Each point presents the mean predicted pK<sub>
<italic>d</italic>
</sub> of each complex from five independent training with respect to experimental determined pK<sub>
<italic>d</italic>
</sub> on training set, validating set, and two test sets of core set v.2016 and core set v.2013.</p>
</caption>
<graphic xlink:href="fchem-09-753002-g003.tif"/>
</fig>
<p>The performance of some published scoring functions and OnionNet-2 tested on CASF-2016 and CASF-2013 are showed in <xref ref-type="fig" rid="F4">Figures 4A,B</xref>, respectively. The corresponding R and RMSE (or SD) achieved by these representative scoring functions can be found in <xref ref-type="sec" rid="s10">Supplementary Table S2</xref>. Firstly, our OnionNet-2 model achieved highest <italic>R</italic> of 0.864 and RMSE of 1.164 with the core set v.2016, and <italic>R</italic>&#x20;&#x3d; 0.821 and RMSE &#x3d; 1.357 with the core set v.2013. These were significantly higher than other scoring functions. The 2<sup>
<italic>nd</italic>
</sup> best scoring function was AGL, which adopted the gradient boosting trees (GBTs) algorithm, focusing on multiscale weighted labeled algebraic subgraphs to characterize protein-ligand interactions <xref ref-type="bibr" rid="B40">Nguyen and Wei (2019)</xref>. For two 3D convolution-based scoring functions <italic>K</italic>
<sub>
<italic>deep</italic>
</sub> <xref ref-type="bibr" rid="B27">Jim&#xe9;nez et&#x20;al. (2018)</xref> and Pafnucy <xref ref-type="bibr" rid="B49">Stepniewska-Dziubinska et&#x20;al. (2018)</xref>, they adopted 3D voxel representation to model the protein-ligand complex and explicitly treated with physical properties of atoms such as hydrophobic, hydrogen-bond donor or acceptor and aromatic etc. into consideration. It is interesting to find that although we only employed the residue-atom contact to mimic the interactions between the protein and the ligand in OnionNet-2, the predicting power is higher. This further reveals that the selected features have a great impact on the predictive power of the CNN-based scoring functions. Secondly, as is expected, the introduction of ML/DL techniques into models has systematically enhanced the predicting accuracy.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Pearson correlation coefficient of different scoring functions on <bold>(A)</bold> CASF-2016 and <bold>(B)</bold> CASF-2013 benchmarks. The scoring functions marked with an asterisk are based on ML/DL models. The purple column is the performance of our proposed&#x20;model.</p>
</caption>
<graphic xlink:href="fchem-09-753002-g004.tif"/>
</fig>
<p>In order to explore the feature importance of different residue-atom combinations to the total performance of the model, we valuated the importance of combination features following our previous strategy. In detail, we re-trained the model with missing a certain residue-atom combination, then calculated the performance change in loss (&#x394;loss). Here, the &#x394;loss is defined as the difference between the loss of a model with missing a certain feature and the loss of the best model. Therefore, larger &#x394;loss represents that this feature has higher importance. The results are summarized in <xref ref-type="fig" rid="F5">Figure&#x20;5</xref>. The most important combination is &#x201c;CYS_H&#x201d;, which is mainly due to the high occurrences of hydrogen bond between CYS and H atom of ligand molecules. Also, &#x201c;CYS_N&#x201d; also showed relatively high importance because N atom of ligand molecule acts as the donor of hydrogen bonds. Besides, &#x201c;ASN_Hal&#x201d; is also recognized to be an important feature to the protein-ligand binding affinity prediction, which may be attributed to the formation of the halogen bond between the halogen atom in the ligand and the O atom in the ASN <xref ref-type="bibr" rid="B6">Cavallo et&#x20;al. (2016)</xref>. Generally, it is worth mention that although we identified the different importance of the combinations, missing of any residue-atom combination indeed does not cause clear decreases in the overall performance of the model. For instance, most &#x394;loss are in the range of 0.02&#x2013;0.04 except that missing of the first 23 combinations caused a &#x394;loss near 0.05, indicating the high robustness of this&#x20;model.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Performance change (&#x394;loss) due to missing features. Performance change (&#x394;loss) when a certain residue-atom combination is removed. The &#x394;loss is defined as the difference between the loss of a model with missing a certain feature and the loss of the best model. The bars indicate the standard deviations of the &#x394;loss for five independent&#x20;runs.</p>
</caption>
<graphic xlink:href="fchem-09-753002-g005.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Evaluation of the Generalization Ability of the Model on Different Test Sets</title>
<p>Generally, DL models display a good generalization behavior in practical applications <xref ref-type="bibr" rid="B39">Neyshabur et&#x20;al. (2017)</xref>. To verify the generalization ability of the OnionNet-2, the CSAR NRC-HiQ data set provided by CSAR. <xref ref-type="bibr" rid="B9">Dunbar et&#x20;al. (2011)</xref> was used as an additional test set in this study. This data set contains two subsets which contain 176 and 167&#x20;protein-ligand complexes, respectively. For the two previous ML models, <italic>K</italic>
<sub>
<italic>deep</italic>
</sub> and RF-Score, the researchers used 55 and 49 complexes in two subsets respectively as test data <xref ref-type="bibr" rid="B27">Jim&#xe9;nez et&#x20;al. (2018)</xref>. To provide a direct comparison with them, we adopted the same data for the OnionNet-2 test. It is worth mention that the two test subsets from the CSAR NRC-HiQ only have two common complexes with core set v.2013, namely 2jdy and 2qmj, and does not overlap with the training set, validation set and core set v.2016. The performance of <italic>K</italic>
<sub>
<italic>deep</italic>
</sub>, RF-Score and OnionNet-2 on these two subsets are shown in <xref ref-type="table" rid="T1">Table&#x20;1</xref>, and the scatter plots of the pK<sub>
<italic>d</italic>
</sub> predicted by OnionNet-2 with respect to experimental pK<sub>
<italic>d</italic>
</sub> can be found in <xref ref-type="sec" rid="s10">Supplementary Figure&#x20;S1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The performance of OnionNet-2, <italic>K</italic>
<sub>
<italic>deep</italic>
</sub> and RF-Score achieved on subsets from CSAR NRC-HiQ data&#x20;set.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left"/>
<th colspan="2" align="center">Subset 1</th>
<th colspan="2" align="center">Subset 2</th>
</tr>
<tr>
<th align="center">R</th>
<th align="center">RMSE</th>
<th align="center">R</th>
<th align="center">RMSE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">OnionNet-2</td>
<td align="char" char=".">0.89</td>
<td align="char" char=".">1.50</td>
<td align="char" char=".">0.87</td>
<td align="char" char=".">1.21</td>
</tr>
<tr>
<td align="left">
<italic>K</italic>
<sub>
<italic>deep</italic>
</sub> <xref ref-type="bibr" rid="B27">Jim&#xe9;nez et&#x20;al. (2018)</xref>
</td>
<td align="char" char=".">0.72</td>
<td align="char" char=".">2.09</td>
<td align="char" char=".">0.65</td>
<td align="char" char=".">1.92</td>
</tr>
<tr>
<td align="left">RF-Score <xref ref-type="bibr" rid="B27">Jim&#xe9;nez et&#x20;al. (2018)</xref>
</td>
<td align="char" char=".">0.78</td>
<td align="char" char=".">1.99</td>
<td align="char" char=".">0.75</td>
<td align="char" char=".">1.66</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As expected, our model achieved a higher performance than <italic>K</italic>
<sub>
<italic>deep</italic>
</sub> and RF-score. For subset 1, the present OnionNet-2 achieved <italic>R</italic> of 0.89, which is considerably higher than that of <italic>K</italic>
<sub>
<italic>deep</italic>
</sub> (0.72) and RF-Score (0.78). This is also true for subset 2. Especially that, the <italic>R</italic> value of <italic>K</italic>
<sub>
<italic>deep</italic>
</sub> model is only 0.65 for subset 2, indicating weak predicting capability on these data. These results effectively demonstrated that OnionNet-2 has a good generalization ability.</p>
</sec>
<sec id="s3-3">
<title>3.3 Evaluations on Subsets of Non-Experimental Decoy Structures</title>
<p>As all the training and validating sets are composed of well-validated native structures in previous studies, it is largely unknown whether the DL method is capable to distinguish &#x201c;bad data&#x201d; that are incorporated in these integrated data sets, for instance, non-native binding poses. To verify this, we tested OnionNet-2 to deal with non-experimental structures (generated by docking programs). Technically, non-native binding poses (called decoys) were generated based on core set v.2016 complexes by AutoDock Vina <xref ref-type="bibr" rid="B53">Trott and Olson (2010)</xref>; <xref ref-type="bibr" rid="B12">Forli et&#x20;al. (2016)</xref>. The detailed information of the generation of decoys can be found in the Supplementary Information. In CASF-2016 benchmark, the similarity between two binding poses is measured by the root-mean-square deviation (RMSD) value. Following previous studies by <xref ref-type="bibr" rid="B3">Allen et&#x20;al. (2014)</xref>, we adopted the Hungarian algorithm to calculate RMSD between decoy ligand and native structure which is implemented in spyrmsd <xref ref-type="bibr" rid="B35">Meli and Biggin (2020)</xref>. The treatment of decoy was as following:<list list-type="simple">
<list-item>
<p>1) For each receptor, up to 20 decoy ligands were generated by AutoDock Vina. The actual number may be less than 20 because of limited size and shape of the binding pocket in the target protein. For each decoy, the RMSD with respect to native structure was calculated.</p>
</list-item>
<list-item>
<p>2) We used 10 RMSD intervals, [0&#xa0;&#xc5;, 2&#xa0;&#xc5;], [2&#xa0;&#xc5;, 3&#xa0;&#xc5;], [3&#xa0;&#xc5;, 4&#xa0;&#xc5;], &#x2026;, [9&#xa0;&#xc5;, 10&#xa0;&#xc5;] and [10&#xa0;&#xc5;:].</p>
</list-item>
<list-item>
<p>3) For all ligands in every interval, we selected the decoy with the smallest RMSD value to put into the corresponding subsets. 4. Ten test subsets containing non-experimental complexes were used for OnionNet-2 training.</p>
</list-item>
</list>
</p>
<p>The predicting accuracy was evaluated by calculating the RMSE between the predicted pK<sub>
<italic>d</italic>
</sub> of the decoy complex and the pK<sub>
<italic>d</italic>
</sub> of the corresponding native receptor-ligand complex which is shown in <xref ref-type="fig" rid="F6">Figure&#x20;6</xref>. It is clear that, the RMSE quickly increased with increasing RMSD. This is expected because decoys with larger RMSD result in more severe change of &#x25b3;<italic>G</italic>. These results reveal that OnionNet-2 can accurately respond to changes of the ligand binding poses and distinguish the native structure.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>RMSE between the predicted pK<sub>
<italic>d</italic>
</sub> of the decoy complex and the pK<sub>
<italic>d</italic>
</sub> of the corresponding native receptor-ligand complex achieved by OnionNet-2 on different protein-decoy complexes subsets. The bars indicate the standard deviations of RMSE for five independent runs.</p>
</caption>
<graphic xlink:href="fchem-09-753002-g006.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>3.4 Effects of Hydrophobic Scale, Buried Solvent-Accessible Area and Excluded Volume Inside the Binding Pockets on the Prediction Accuracy</title>
<p>Principally, the physical interactions between protein and ligand determine the &#x25b3;<italic>G</italic>. The dominating factors for overall &#x25b3;<italic>G</italic> involve electrostatic interactions, van der Waals interactions, hydrogen bonds, hydration/de-hydration during complexation. However, such mechanistic interactions were not directly input into DL features. At molecular level, these involves the size and shape of the binding pocket, and the nature of residues around the binding pocket which determine its physicochemical characteristics <xref ref-type="bibr" rid="B48">Stank et&#x20;al. (2016)</xref>. However, whether DL models can accurately represent the structural specificity of the binding pocket is poorly documented.</p>
<p>The entire CASF-2016 test set can be divided into three subsets by each of three descriptors according to physical classifications of the binding pocket on the target protein <xref ref-type="bibr" rid="B50">Su et&#x20;al. (2018)</xref>. The three descriptors include H-scale (hydrophobic scale of the binding pocket), &#x25b3;<italic>SAS</italic> (buried percentage of the solvent-accessible area of the ligand after binding) and &#x25b3;<italic>VOL</italic> (excluded volume inside the binding pocket after ligand binding). Protein-ligand complexes in CASF-2016 were grouped into 57 clusters, and the authors sorted all 57 clusters in ascending order by each descriptor. Then, these complex clusters were divided into three subsets according to the chosen descriptor, labeled as H1, H2, and H3 or S1, S2, and S3 or V1, V2, and V3. These subsets were used as validations of our OnionNet-2 model. As comparison, previous scoring functions were also tested on these three sets of subsets by <xref ref-type="bibr" rid="B50">Su et&#x20;al. (2018)</xref>, and the results are summarized in <xref ref-type="table" rid="T2">Table&#x20;2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Pearson correlation coefficients achieved by OnionNet-2 and previous scoring functions on three series of subsets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left"/>
<th colspan="3" align="center">Subset H</th>
<th colspan="3" align="center">Subset S</th>
<th colspan="3" align="center">Subset V</th>
</tr>
<tr>
<th align="center">H1</th>
<th align="center">H2</th>
<th align="center">H3</th>
<th align="center">S1</th>
<th align="center">S2</th>
<th align="center">S3</th>
<th align="center">V1</th>
<th align="center">V2</th>
<th align="center">V3</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">OnionNet-2</td>
<td align="char" char=".">0.872</td>
<td align="char" char=".">0.866</td>
<td align="char" char=".">0.856</td>
<td align="char" char=".">0.868</td>
<td align="char" char=".">0.839</td>
<td align="char" char=".">0.869</td>
<td align="char" char=".">0.856</td>
<td align="char" char=".">0.774</td>
<td align="char" char=".">0.866</td>
</tr>
<tr>
<td align="left">&#x25b3;<sub>
<italic>Vina</italic>
</sub>RF<sub>20</sub>
</td>
<td align="char" char=".">0.820</td>
<td align="char" char=".">0.832</td>
<td align="char" char=".">0.804</td>
<td align="char" char=".">0.843</td>
<td align="char" char=".">0.765</td>
<td align="char" char=".">0.823</td>
<td align="char" char=".">0.727</td>
<td align="char" char=".">0.760</td>
<td align="char" char=".">0.818</td>
</tr>
<tr>
<td align="left">X-Score</td>
<td align="char" char=".">0.698</td>
<td align="char" char=".">0.570</td>
<td align="char" char=".">0.661</td>
<td align="char" char=".">0.743</td>
<td align="char" char=".">0.536</td>
<td align="char" char=".">0.572</td>
<td align="char" char=".">0.437</td>
<td align="char" char=".">0.622</td>
<td align="char" char=".">0.579</td>
</tr>
<tr>
<td align="left">X-Score<sup>
<italic>HS</italic>
</sup>
</td>
<td align="char" char=".">0.711</td>
<td align="char" char=".">0.565</td>
<td align="char" char=".">0.647</td>
<td align="char" char=".">0.748</td>
<td align="char" char=".">0.557</td>
<td align="char" char=".">0.546</td>
<td align="char" char=".">0.433</td>
<td align="char" char=".">0.630</td>
<td align="char" char=".">0.559</td>
</tr>
<tr>
<td align="left">&#x25b3;SAS</td>
<td align="char" char=".">0.641</td>
<td align="char" char=".">0.643</td>
<td align="char" char=".">0.589</td>
<td align="char" char=".">0.746</td>
<td align="char" char=".">0.572</td>
<td align="char" char=".">0.494</td>
<td align="char" char=".">0.480</td>
<td align="char" char=".">0.669</td>
<td align="char" char=".">0.541</td>
</tr>
<tr>
<td align="left">X-Score<sup>
<italic>HP</italic>
</sup>
</td>
<td align="char" char=".">0.669</td>
<td align="char" char=".">0.558</td>
<td align="char" char=".">0.672</td>
<td align="char" char=".">0.740</td>
<td align="char" char=".">0.510</td>
<td align="char" char=".">0.575</td>
<td align="char" char=".">0.450</td>
<td align="char" char=".">0.575</td>
<td align="char" char=".">0.580</td>
</tr>
<tr>
<td align="left">AutoDock Vina</td>
<td align="char" char=".">0.626</td>
<td align="char" char=".">0.586</td>
<td align="char" char=".">0.641</td>
<td align="char" char=".">0.745</td>
<td align="char" char=".">0.521</td>
<td align="char" char=".">0.507</td>
<td align="char" char=".">0.484</td>
<td align="char" char=".">0.564</td>
<td align="char" char=".">0.486</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Results of the last six rows taken from <xref ref-type="bibr" rid="B50">Su et&#x20;al. (2018)</xref>.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>As can be seen in <xref ref-type="table" rid="T2">Table&#x20;2</xref>, OnionNet-2 achieved higher prediction accuracy compared with other soring functions when tested on H, S, and V-series subsets. This indicates that the feature based on the contact number of residue-atom pairs in multiple shell is capable of capturing the hydrophobic scale of the binding pocket. The number of contacts in different shells (specifically the shells within the binding pocket) may be able to reflect the buried solvent-accessible surface area and the excluded volume of the ligand.</p>
<p>We noticed that, compared to other subsets, the R value of OnionNet-2 on V2 subset is clearly lower than other subsets (nevertheless it is still high than other scoring functions). This may indicate that our model is less sensitive to medium-sized binding pockets. Thus it may be still challenging for current scoring functions to recognize the size and shape of the binding pocket.</p>
<p>Furthermore, we plotted the detailed scatter plots of predicted pK<sub>
<italic>d</italic>
</sub> and experimental pK<sub>
<italic>d</italic>
</sub> in <xref ref-type="sec" rid="s10">Supplementary Figure S3</xref> according to the specific H, S and V range. It is interesting to find almost no dependence of pK<sub>
<italic>d</italic>
</sub> with the values of H, S, or V. Thus we speculate that a more realistic descriptor for the ligand characteristic in the binding pocket is essential to guide the protein-ligand &#x25b3;<italic>G</italic> prediction.</p>
</sec>
<sec id="s3-5">
<title>3.5 Discussions of Screening Power</title>
<p>From the above results, we demonstrated that OnionNet-2 has high efficiency in treating with protein-ligand binding affinity prediction with simple calculations of structural features. However, we noticed a work from Hou group reporting that mostly developed ML models (including the OnionNet) performed poorly in the virtual screening tasks <xref ref-type="bibr" rid="B45">Shen et&#x20;al. (2020b)</xref>. In such screening power examinations, theoretical models have to pick true binders from a lot of false &#x201c;decoy&#x201d; molecules. It is not surprising that all ML models which were trained soly on true binders (for example, PDBbind databases) were not taught to distinguish decoy molecules from the true binders. In order to improve the screening power of the ML/DL based scoring functions, decoy molecules should be included in the training sets. Such work is now being undertaken by&#x20;us.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Conclusion</title>
<p>To summarize, a 2D convolution-based CNN model, OnionNet-2, is proposed for prediction of the protein-ligand binding free energy. The contacting pair numbers between the protein residues and the ligand atoms were used as features for DL training. Using CASF-2013 and CASF-2016 as benchmarks, our model achieved the highest accuracy to predict &#x25b3;<italic>G</italic> than previous scoring functions. In addition, when employing different versions of PDBbind database for training, the performance of OnionNet-2 is nearly the same. We also evaluated the generalization ability of the model through testing on the CSAR NRC-HiQ data set and the decoys structures. Our result also indicates that OnionNet-2 has the capability to recognize these physical natures (in detail, hydrophobic scale of the binding pocket, buried percentage of the solvent-accessible area of the ligand upon binding and excluded volume inside the binding pocket upon ligand binding) of the ligand-binding pocket interaction. This systematic study also verified our initial hypothesis that is using the intrinsic types residues as the basic units can better characterize the physicochemical properties of the protein, which will be beneficial to improve the performance of the protein-ligand binding affinity prediction.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s10">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>ZW performed experiments, analyzed data and wrote the article. LZ, WL, and YM designed the experiment and analyzed data. YL, YQ, Y-QL, and MZ analyzed data. ZW, LZ, WL, and YM wrote the article.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This work is supported by the Natural Science Foundation of Shandong Province (ZR2020JQ04), National Natural Science Foundation of China (11874238) and Singapore MOE Tier 1 Grant RG146/17. This work is also supported by the Ministry of Education, Singapore, under its Academic Research Fund Tier 2, MOE-T2EP30120-0007.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fchem.2021.753002/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fchem.2021.753002/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Abadi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Barham</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Davis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dean</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). &#x201c;<article-title>Tensorflow: A System for Large-Scale Machine Learning</article-title>,&#x201d; in <conf-name>12th {USENIX} symposium on operating systems design and implementation ({OSDI} 16)</conf-name>, <fpage>265</fpage>&#x2013;<lpage>283</lpage>. </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ain</surname>
<given-names>Q. U.</given-names>
</name>
<name>
<surname>Aleksandrova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Roessler</surname>
<given-names>F. D.</given-names>
</name>
<name>
<surname>Ballester</surname>
<given-names>P. J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Machine-learning Scoring Functions to Improve Structure-Based Binding Affinity Prediction and Virtual Screening</article-title>. <source>Wires Comput. Mol. Sci.</source> <volume>5</volume>, <fpage>405</fpage>&#x2013;<lpage>424</lpage>. <pub-id pub-id-type="doi">10.1002/wcms.1225</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Allen</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Rizzo</surname>
<given-names>R. C.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Implementation of the Hungarian Algorithm to Account for Ligand Symmetry and Similarity in Structure-Based Design</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>54</volume>, <fpage>518</fpage>&#x2013;<lpage>529</lpage>. <pub-id pub-id-type="doi">10.1021/ci400534h</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Angermueller</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>P&#xe4;rnamaa</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Parts</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Stegle</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Deep Learning for Computational Biology</article-title>. <source>Mol. Syst. Biol.</source> <volume>12</volume>, <fpage>878</fpage>. <pub-id pub-id-type="doi">10.15252/msb.20156651</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ballester</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Mitchell</surname>
<given-names>J.&#x20;B. O.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>A Machine Learning Approach to Predicting Protein-Ligand Binding Affinity with Applications to Molecular Docking</article-title>. <source>Bioinformatics</source> <volume>26</volume>, <fpage>1169</fpage>&#x2013;<lpage>1175</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq112</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cavallo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Metrangolo</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Milani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pilati</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Priimagi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Resnati</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>The Halogen Bond</article-title>. <source>Chem. Rev.</source> <volume>116</volume>, <fpage>2478</fpage>&#x2013;<lpage>2601</lpage>. <pub-id pub-id-type="doi">10.1021/acs.chemrev.5b00484</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Engkvist</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Olivecrona</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Blaschke</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The Rise of Deep Learning in Drug Discovery</article-title>. <source>Drug Discov. Today</source> <volume>23</volume>, <fpage>1241</fpage>&#x2013;<lpage>1250</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2018.01.039</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Du</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>Y.-L.</given-names>
</name>
<name>
<surname>Ai</surname>
<given-names>S.-M.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sang</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Insights into Protein-Ligand Interactions: Mechanisms, Models, and Methods</article-title>. <source>Ijms</source> <volume>17</volume>, <fpage>144</fpage>. <pub-id pub-id-type="doi">10.3390/ijms17020144</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dunbar</surname>
<given-names>J.&#x20;B.</given-names>
<suffix>Jr</suffix>
</name>
<name>
<surname>Smith</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.-Y.</given-names>
</name>
<name>
<surname>Ung</surname>
<given-names>P. M.-U.</given-names>
</name>
<name>
<surname>Lexa</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Khazanov</surname>
<given-names>N. A.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>CSAR Benchmark Exercise of 2010: Selection of the Protein-Ligand Complexes</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>51</volume>, <fpage>2036</fpage>&#x2013;<lpage>2046</lpage>. <pub-id pub-id-type="doi">10.1021/ci200082t</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Durrant</surname>
<given-names>J.&#x20;D.</given-names>
</name>
<name>
<surname>McCammon</surname>
<given-names>J.&#x20;A.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>NNScore: A Neural-Network-Based Scoring Function for the Characterization of Protein&#x2212;Ligand Complexes</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>50</volume>, <fpage>1865</fpage>&#x2013;<lpage>1871</lpage>. <pub-id pub-id-type="doi">10.1021/ci100244v</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ellingson</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Davis</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Allen</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Machine Learning and Ligand Binding Predictions: a Review of Data, Methods, and Obstacles</article-title>. <source>Biochim. Biophys. Acta (Bba) - Gen. Subjects</source> <volume>1864</volume>, <fpage>129545</fpage>. <pub-id pub-id-type="doi">10.1016/j.bbagen.2020.129545</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Forli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huey</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pique</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Sanner</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Goodsell</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Olson</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Computational Protein-Ligand Docking and Virtual Drug Screening with the AutoDock Suite</article-title>. <source>Nat. Protoc.</source> <volume>11</volume>, <fpage>905</fpage>&#x2013;<lpage>919</lpage>. <pub-id pub-id-type="doi">10.1038/nprot.2016.051</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garc&#xed;a-Sosa</surname>
<given-names>A. T.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Hydration Properties of Ligands and Drugs in Protein Binding Sites: Tightly-Bound, Bridging Water Molecules and Their Effects and Consequences on Molecular Design Strategies</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>53</volume>, <fpage>1388</fpage>&#x2013;<lpage>1405</lpage>. <pub-id pub-id-type="doi">10.1021/ci3005786</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gawehn</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hiss</surname>
<given-names>J.&#x20;A.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Deep Learning in Drug Discovery</article-title>. <source>Mol. Inf.</source> <volume>35</volume>, <fpage>3</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1002/minf.201501008</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghasemi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Mehridehnavi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>P&#xe9;rez-Garrido</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>P&#xe9;rez-S&#xe1;nchez</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Neural Network and Deep-Learning Algorithms Used in Qsar Studies: Merits and Drawbacks</article-title>. <source>Drug Discov. Today</source> <volume>23</volume>, <fpage>1784</fpage>&#x2013;<lpage>1790</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2018.06.016</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gilson</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.-X.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Calculation of Protein-Ligand Binding Affinities</article-title>. <source>Annu. Rev. Biophys. Biomol. Struct.</source> <volume>36</volume>, <fpage>21</fpage>&#x2013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1146/annurev.biophys.36.040306.132550</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gomes</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ramsundar</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Feinberg</surname>
<given-names>E. N.</given-names>
</name>
<name>
<surname>Pande</surname>
<given-names>V. S.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Atomic Convolutional Networks for Predicting Protein-Ligand Binding Affinity</source>. <comment>arXiv preprint arXiv:1703.10603</comment>. </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grinter</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Challenges, Applications, and Recent Advances of Protein-Ligand Docking in Structure-Based Drug Design</article-title>. <source>Molecules</source> <volume>19</volume>, <fpage>10150</fpage>&#x2013;<lpage>10176</lpage>. <pub-id pub-id-type="doi">10.3390/molecules190710150</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guedes</surname>
<given-names>I. A.</given-names>
</name>
<name>
<surname>Pereira</surname>
<given-names>F. S. S.</given-names>
</name>
<name>
<surname>Dardenne</surname>
<given-names>L. E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Empirical Scoring Functions for Structure-Based Virtual Screening: Applications, Critical Aspects, and Challenges</article-title>. <source>Front. Pharmacol.</source> <volume>9</volume>, <fpage>1089</fpage>. <pub-id pub-id-type="doi">10.3389/fphar.2018.01089</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guvench</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>MacKerell</surname>
<given-names>A. D.</given-names>
<suffix>Jr</suffix>
</name>
</person-group> (<year>2009</year>). <article-title>Computational Evaluation of Protein-Small Molecule Binding</article-title>. <source>Curr. Opin. Struct. Biol.</source> <volume>19</volume>, <fpage>56</fpage>&#x2013;<lpage>61</lpage>. <pub-id pub-id-type="doi">10.1016/j.sbi.2008.11.009</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hansen</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Van Gunsteren</surname>
<given-names>W. F.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Practical Aspects of Free-Energy Calculations: a Review</article-title>. <source>J.&#x20;Chem. Theor. Comput.</source> <volume>10</volume>, <fpage>2632</fpage>&#x2013;<lpage>2647</lpage>. <pub-id pub-id-type="doi">10.1021/ct500161f</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heck</surname>
<given-names>G. S.</given-names>
</name>
<name>
<surname>Pintro</surname>
<given-names>V. O.</given-names>
</name>
<name>
<surname>Pereira</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>de &#xc1;vila</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Levin</surname>
<given-names>N. M. B.</given-names>
</name>
<name>
<surname>de Azevedo</surname>
<given-names>W. F.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Supervised Machine Learning Methods Applied to Predict Ligand- Binding Affinity</article-title>. <source>Curr. Med. Chem.</source> <volume>24</volume>, <fpage>2459</fpage>&#x2013;<lpage>2470</lpage>. <pub-id pub-id-type="doi">10.2174/0929867324666170623092503</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kalyanaraman</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Irwin</surname>
<given-names>J.&#x20;J.</given-names>
</name>
<name>
<surname>Jacobson</surname>
<given-names>M. P.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Physics-Based Scoring of Protein&#x2212;Ligand Complexes: Enrichment of Known Inhibitors in Large-Scale Virtual Screening</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>46</volume>, <fpage>243</fpage>&#x2013;<lpage>253</lpage>. <pub-id pub-id-type="doi">10.1021/ci0502855</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>S.-Y.</given-names>
</name>
<name>
<surname>Grinter</surname>
<given-names>S. Z.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Scoring Functions and Their Evaluation Methods for Protein-Ligand Docking: Recent Advances and Future Directions</article-title>. <source>Phys. Chem. Chem. Phys.</source> <volume>12</volume>, <fpage>12899</fpage>&#x2013;<lpage>12908</lpage>. <pub-id pub-id-type="doi">10.1039/c0cp00151a</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huey</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Morris</surname>
<given-names>G. M.</given-names>
</name>
<name>
<surname>Olson</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Goodsell</surname>
<given-names>D. S.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>A Semiempirical Free Energy Force Field with Charge-Based Desolvation</article-title>. <source>J.&#x20;Comput. Chem.</source> <volume>28</volume>, <fpage>1145</fpage>&#x2013;<lpage>1152</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.20634</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jim&#xe9;nez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Doerr</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mart&#xed;nez-Rosell</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rose</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>De Fabritiis</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Deepsite: Protein-Binding Site Predictor Using 3d-Convolutional Neural Networks</article-title>. <source>Bioinformatics</source> <volume>33</volume>, <fpage>3036</fpage>&#x2013;<lpage>3042</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx350</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jim&#xe9;nez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>&#x160;kali&#x10d;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mart&#xed;nez-Rosell</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>De Fabritiis</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>KDEEP: Protein-Ligand Absolute Binding Affinity Prediction via 3D-Convolutional Neural Networks</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>58</volume>, <fpage>287</fpage>&#x2013;<lpage>296</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.7b00650</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lavecchia</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep Learning in Drug Discovery: Opportunities, Challenges and Future Prospects</article-title>. <source>Drug Discov. Today</source> <volume>24</volume>, <fpage>2017</fpage>&#x2013;<lpage>2032</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2019.07.006</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lavecchia</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Machine-learning Approaches in Drug Discovery: Methods and Applications</article-title>. <source>Drug Discov. Today</source> <volume>20</volume>, <fpage>318</fpage>&#x2013;<lpage>331</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2014.10.012</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2014a</year>). <article-title>Comparative Assessment of Scoring Functions on an Updated Benchmark: 2. Evaluation Methods and General Results</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>54</volume>, <fpage>1717</fpage>&#x2013;<lpage>1736</lpage>. <pub-id pub-id-type="doi">10.1021/ci500081m</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2014b</year>). <article-title>Comparative Assessment of Scoring Functions on an Updated Benchmark: 1. Compilation of the Test Set</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>54</volume>, <fpage>1700</fpage>&#x2013;<lpage>1716</lpage>. <pub-id pub-id-type="doi">10.1021/ci500080q</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Assessing Protein-Ligand Interaction Scoring Functions with the CASF-2013 Benchmark</article-title>. <source>Nat. Protoc.</source> <volume>13</volume>, <fpage>666</fpage>&#x2013;<lpage>680</lpage>. <pub-id pub-id-type="doi">10.1038/nprot.2017.114</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Classification of Current Scoring Functions</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>55</volume>, <fpage>475</fpage>&#x2013;<lpage>482</lpage>. <pub-id pub-id-type="doi">10.1021/ci500731a</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lo</surname>
<given-names>Y.-C.</given-names>
</name>
<name>
<surname>Rensi</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Torng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Altman</surname>
<given-names>R. B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Machine Learning in Chemoinformatics and Drug Discovery</article-title>. <source>Drug Discov. Today</source> <volume>23</volume>, <fpage>1538</fpage>&#x2013;<lpage>1546</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2018.05.010</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meli</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Biggin</surname>
<given-names>P. C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Spyrmsd: Symmetry-Corrected Rmsd Calculations in python</article-title>. <source>J.&#x20;Cheminform.</source> <volume>12</volume>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1186/s13321-020-00455-2</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Michel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Essex</surname>
<given-names>J.&#x20;W.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Prediction of Protein-Ligand Binding Affinity by Free Energy Simulations: Assumptions, Pitfalls and Expectations</article-title>. <source>J.&#x20;Comput. Aided Mol. Des.</source> <volume>24</volume>, <fpage>639</fpage>&#x2013;<lpage>658</lpage>. <pub-id pub-id-type="doi">10.1007/s10822-010-9363-3</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morris</surname>
<given-names>G. M.</given-names>
</name>
<name>
<surname>Goodsell</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Halliday</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Huey</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hart</surname>
<given-names>W. E.</given-names>
</name>
<name>
<surname>Belew</surname>
<given-names>R. K.</given-names>
</name>
<etal/>
</person-group> (<year>1998</year>). <article-title>Automated Docking Using a Lamarckian Genetic Algorithm and an Empirical Binding Free Energy Function</article-title>. <source>J.&#x20;Comput. Chem.</source> <volume>19</volume>, <fpage>1639</fpage>&#x2013;<lpage>1662</lpage>. <pub-id pub-id-type="doi">10.1002/(sici)1096-987x(19981115)19:14&#x3c;1639:aid-jcc10&#x3e;3.0.co;2-b</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morrone</surname>
<given-names>J.&#x20;A.</given-names>
</name>
<name>
<surname>Weber</surname>
<given-names>J.&#x20;K.</given-names>
</name>
<name>
<surname>Huynh</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cornell</surname>
<given-names>W. D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Combining Docking Pose Rank and Structure with Deep Learning Improves Protein-Ligand Binding Mode Prediction over a Baseline Docking Approach</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>60</volume>, <fpage>4170</fpage>&#x2013;<lpage>4179</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.9b00927</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Neyshabur</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bhojanapalli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>McAllester</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Srebro</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Exploring Generalization in Deep Learning</source>. <comment>arXiv preprint arXiv:1706.08947</comment>. </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>D. D.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>G.-W.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>AGL-score: Algebraic Graph Learning Score for Protein-Ligand Binding Scoring, Ranking, Docking, and Screening</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>59</volume>, <fpage>3291</fpage>&#x2013;<lpage>3304</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.9b00334</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#xd6;zt&#xfc;rk</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>&#xd6;zg&#xfc;r</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ozkirimli</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>DeepDTA: Deep Drug-Target Binding Affinity Prediction</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>i821</fpage>&#x2013;<lpage>i829</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty593</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Pak</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>A Review of Deep Learning in Image Recognition</article-title>,&#x201d; in <conf-name>2017 4th international conference on computer applications and information processing technology (CAIPT)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>3</lpage>. <pub-id pub-id-type="doi">10.1109/caipt.2017.8320684</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedregosa</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Varoquaux</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gramfort</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Michel</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Thirion</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Grisel</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Scikit-learn: Machine Learning in python</article-title>. <source>J.&#x20;Machine Learn. Res.</source> <volume>12</volume>, <fpage>2825</fpage>&#x2013;<lpage>2830</lpage>. </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>From Machine Learning to Deep Learning: Advances in Scoring Functions for Protein&#x2013;Ligand Docking</article-title>. <source>Wiley Interdiscip. Rev. Comput. Mol. Sci.</source> <volume>10</volume>, <fpage>e1429</fpage>. <pub-id pub-id-type="doi">10.1002/wcms.1429</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2020b</year>). <article-title>Beware of the Generic Machine Learning-Based Scoring Functions in Structure-Based Virtual Screening</article-title>. <source>Brief. Bioinform.</source> <volume>22</volume>, <fpage>bbaa070</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa070</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Se-onionnet: A Convolution Neural Network for Protein-Ligand Binding Affinity Prediction</article-title>. <source>Front. Genet.</source> <volume>11</volume>, <fpage>1805</fpage>. </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Bayden</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>The Roles of Water in the Protein Matrix: a Largely Untapped Resource for Drug Discovery</article-title>. <source>J.&#x20;Med. Chem.</source> <volume>60</volume>, <fpage>6781</fpage>&#x2013;<lpage>6827</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jmedchem.7b00057</pub-id> </citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stank</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kokh</surname>
<given-names>D. B.</given-names>
</name>
<name>
<surname>Fuller</surname>
<given-names>J.&#x20;C.</given-names>
</name>
<name>
<surname>Wade</surname>
<given-names>R. C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Protein Binding Pocket Dynamics</article-title>. <source>Acc. Chem. Res.</source> <volume>49</volume>, <fpage>809</fpage>&#x2013;<lpage>815</lpage>. <pub-id pub-id-type="doi">10.1021/acs.accounts.5b00516</pub-id> </citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stepniewska-Dziubinska</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Zielenkiewicz</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Siedlecki</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Development and Evaluation of a Deep Learning Model for Protein-Ligand Binding Affinity Prediction</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>3666</fpage>&#x2013;<lpage>3674</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty374</pub-id>, </citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Comparative Assessment of Scoring Functions: the Casf-2016 Update</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>59</volume>, <fpage>895</fpage>&#x2013;<lpage>913</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.8b00545</pub-id> </citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Torng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Altman</surname>
<given-names>R. B.</given-names>
</name>
</person-group> (<year>2019a</year>). <article-title>Graph Convolutional Neural Networks for Predicting Drug-Target Interactions</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>59</volume>, <fpage>4131</fpage>&#x2013;<lpage>4149</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.9b00628</pub-id> </citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Torng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Altman</surname>
<given-names>R. B.</given-names>
</name>
</person-group> (<year>2019b</year>). <article-title>High Precision Protein Functional Site Detection Using 3d Convolutional Neural Networks</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>1503</fpage>&#x2013;<lpage>1512</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty813</pub-id> </citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Trott</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Olson</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Autodock Vina: Improving the Speed and Accuracy of Docking with a New Scoring Function, Efficient Optimization, and Multithreading</article-title>. <source>J.&#x20;Comput. Chem.</source> <volume>31</volume>, <fpage>455</fpage>&#x2013;<lpage>461</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.21334</pub-id> </citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vamathevan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Czodrowski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Dunham</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Ferran</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Applications of Machine Learning in Drug Discovery and Development</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>18</volume>, <fpage>463</fpage>&#x2013;<lpage>477</lpage>. <pub-id pub-id-type="doi">10.1038/s41573-019-0024-5</pub-id> </citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lai</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Further Development and Validation of Empirical Scoring Functions for Structure-Based Binding Affinity Prediction</article-title>. <source>J.&#x20;Computer-Aided Mol. Des.</source> <volume>16</volume>, <fpage>11</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1023/a:1016357811882</pub-id> </citation>
</ref>
<ref id="B56">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Deep Learning in Bioinformatics</source>. </citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Winter</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Montanari</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>No&#xe9;</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Clevert</surname>
<given-names>D.-A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Learning Continuous and Data-Driven Molecular Descriptors by Translating Equivalent Chemical Representations</article-title>. <source>Chem. Sci.</source> <volume>10</volume>, <fpage>1692</fpage>&#x2013;<lpage>1701</lpage>. <pub-id pub-id-type="doi">10.1039/c8sc04175j</pub-id> </citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Swanson</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Coley</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Eiden</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Analyzing Learned Molecular Representations for Property Prediction</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>59</volume>, <fpage>3370</fpage>&#x2013;<lpage>3388</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.9b00237</pub-id> </citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>An Improved Pmf Scoring Function for Universally Predicting the Interactions of a Ligand with Protein, Dna, and Rna</article-title>. <source>J.&#x20;Chem. Inf. Model.</source> <volume>48</volume>, <fpage>1438</fpage>&#x2013;<lpage>1447</lpage>. <pub-id pub-id-type="doi">10.1021/ci7004719</pub-id> </citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>OnionNet: a Multiple-Layer Intermolecular-Contact-Based Convolutional Neural Network for Protein-Ligand Binding Affinity Prediction</article-title>. <source>ACS Omega</source> <volume>4</volume>, <fpage>15956</fpage>&#x2013;<lpage>15965</lpage>. <pub-id pub-id-type="doi">10.1021/acsomega.9b01997</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>