<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="review-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mol. Biosci.</journal-id>
<journal-title>Frontiers in Molecular Biosciences</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mol. Biosci.</abbrev-journal-title>
<issn pub-type="epub">2296-889X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1194962</article-id>
<article-id pub-id-type="doi">10.3389/fmolb.2023.1194962</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Molecular Biosciences</subject>
<subj-group>
<subject>Review</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>HINT, a code for understanding the interaction between biomolecules: a tribute to Donald J. Abraham</article-title>
<alt-title alt-title-type="left-running-head">Kellogg et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fmolb.2023.1194962">10.3389/fmolb.2023.1194962</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kellogg</surname>
<given-names>Glen E.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/180856/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Marabotti</surname>
<given-names>Anna</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/132039/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Spyrakis</surname>
<given-names>Francesca</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/651241/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Mozzarelli</surname>
<given-names>Andrea</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/142943/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Medicinal Chemistry and Institute for Structural Biology</institution>, <institution>Drug Discovery and Development</institution>, <institution>Virginia Commonwealth University</institution>, <addr-line>Richmond</addr-line>, <addr-line>VA</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Chemistry and Biology &#x201c;A Zambelli&#x201d;</institution>, <institution>University of Salerno</institution>, <addr-line>Fisciano (SA)</addr-line>, <country>Italy</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Drug Science and Technology</institution>, <institution>University of Turin</institution>, <addr-line>Turin</addr-line>, <country>Italy</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Food and Drug</institution>, <institution>University of Parma and Institute of Biophysics</institution>, <addr-line>Parma</addr-line>, <country>Italy</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1526839/overview">Gloria C. Ferreira</ext-link>, University of South Florida, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1518170/overview">Marcus Fischer</ext-link>, St. Jude Children&#x2019;s Research Hospital, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1611052/overview">Traian Sulea</ext-link>, National Research Council Canada (NRC), Canada</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Glen E. Kellogg, <email>glen.kellogg@vcu.edu</email>; Andrea Mozzarelli, <email>andrea.mozzarelli@unipr.it</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>06</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>10</volume>
<elocation-id>1194962</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>03</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>05</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Kellogg, Marabotti, Spyrakis and Mozzarelli.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Kellogg, Marabotti, Spyrakis and Mozzarelli</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>A long-lasting goal of computational biochemists, medicinal chemists, and structural biologists has been the development of tools capable of deciphering the molecule&#x2013;molecule interaction code that produces a rich variety of complex biomolecular assemblies comprised of the many different simple and biological molecules of life: water, small metabolites, cofactors, substrates, proteins, DNAs, and RNAs. Software applications that can mimic the interactions amongst all of these species, taking account of the laws of thermodynamics, would help gain information for understanding qualitatively and quantitatively key determinants contributing to the energetics of the bimolecular recognition process. This, in turn, would allow the design of novel compounds that might bind at the intermolecular interface by either preventing or reinforcing the recognition. HINT, hydropathic interaction, was a model and software code developed from a deceptively simple idea of Donald Abraham with the close collaboration with Glen Kellogg at Virginia Commonwealth University. HINT is based on a function that scores atom&#x2013;atom interaction using LogP, the partition coefficient of any molecule between two phases; here, the solvents are water that mimics the cytoplasm milieu and octanol that mimics the protein internal hydropathic environment. This review summarizes the results of the extensive and successful collaboration between Abraham and Kellogg at VCU and the group at the University of Parma for testing HINT in a variety of different biomolecular interactions, from proteins with ligands to proteins with DNA.</p>
</abstract>
<kwd-group>
<kwd>hydrophatic interactions</kwd>
<kwd>LogP</kwd>
<kwd>HINT</kwd>
<kwd>protein&#x2013;ligand</kwd>
<kwd>protein&#x2013;protein</kwd>
<kwd>protein&#x2013;DNA complexes</kwd>
<kwd>water thermodynamics</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Protein Biochemistry for Basic and Applied Sciences</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Proteins play many functions in living systems as carriers, enzymes, antibodies, receptors, hormones, mechanical support, and storage. In essentially all these functions, proteins interact with small molecules, metals, and other proteins or peptides. Whenever an interaction occurs between two or more molecules, the recognition is dictated by three key elements: i) complementarity in shape, ii) complementarity in interacting moieties, and iii) free energy of binding. There is also a fourth element that deals with time, i.e., how rapidly molecules see each other and how long they stay together. In some cases, this is a few milliseconds, and in other cases, they remain associated for their entire life before degradation.</p>
<p>Recognition is based on structural and chemical complementarity (<xref ref-type="fig" rid="F1">Figures 1A,B</xref>). Structural complementarity might be obtained upon induced fit (<xref ref-type="bibr" rid="B49">Koshland, 1958</xref>; <xref ref-type="bibr" rid="B23">Cozzini et al., 2008</xref>) or via a selection among an ensemble of conformations characterized by small energetic differences (<xref ref-type="bibr" rid="B53">Ma et al., 1999</xref>). Before and after an interaction, the energy landscape is modified, favoring the conformation that better matches the interacting pair of molecules. This process is driven by energy contributions arising from the determinants at the interface of the partner molecules. Ionic&#x2013;ionic, polar&#x2013;ionic, polar&#x2013;polar, and apolar&#x2013;apolar hydrogen bonds and hydrophobic interactions contribute to the overall energetics of the recognition. The degree of energetic contributions depends on geometric parameters, such as distance, and the dielectric constant of the medium where interacting groups are localized. Thus, a detailed and precise prediction of the affinity between two molecules can most confidently be obtained when the three-dimensional structure of the complex is known at high resolution and when a quantum mechanical analysis has generated a complete electronic description of the environment. This is a quite challenging goal that has yet to be met, except in toy systems (<xref ref-type="bibr" rid="B72">Yilmazer and Korth, 2016</xref>; <xref ref-type="bibr" rid="B19">Cavasotto and Aucar, 2020</xref>; <xref ref-type="bibr" rid="B47">Kirsopp et al., 2021</xref>). While robust and accurate prediction of the strength of the interaction is complex, the experimental determination of the affinity is often easier but can lack the context of &#x201c;visualizing&#x201d; the specific interactions involved. However, for many purposes, the capability of predicting affinities of a complex might actually direct experimental work, such as in the development of potential drugs via structure-based or ligand-based methods. Many efforts have been devoted to the prediction of affinities between proteins and small ligands, proteins, or nucleic acids, and many thoughtful reviews have been published (<xref ref-type="bibr" rid="B8">Ajay and Murcko, 1995</xref>; <xref ref-type="bibr" rid="B21">Cozzini et al., 2004</xref>; <xref ref-type="bibr" rid="B27">Foloppe and Hubbard, 2006</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Schematic representation of the molecular events that take place when a ligand binds to a protein site: <bold>(a)</bold> several interactions between ligand polar and apolar groups and protein sidechain residues are formed, and <bold>(b)</bold> water molecules within the active site and bound to the ligand are released into the solvent. These interactions determine the strength of the protein&#x2013;ligand complex and are computationally evaluated by HINT. <bold>(C)</bold> The toolboxes of HINT for the evaluation of ligand&#x2013;protein interactions, including water molecules.</p>
</caption>
<graphic xlink:href="fmolb-10-1194962-g001.tif"/>
</fig>
<p>Here, we will focus on a very simple approach that exploits both experimental and computational information to generate a score of the interaction that is directly related to the free energy of binding (<xref ref-type="fig" rid="F1">Figure 1C</xref>). This approach was called HINT, which stands for hydropathic interactions (<xref ref-type="bibr" rid="B40">Kellogg and Abraham, 2000</xref>). It was developed by a collaboration between the inspired medicinal chemist Donald Abraham and his talented colleague Glen Kellogg at Virginia Commonwealth University (VCU). Abraham had a long-standing interest in the &#x201c;hydrophobic effect&#x201d; and its importance in quantitative structure&#x2013;activity relationship (QSAR) and protein structure. He had, in fact, collaborated with Al Leo of Pomona College in an early article attempting to unify the understanding of LogP (and hydrophobicity) between medicinal chemists and those structural biologists predicting protein secondary structure based on sidechain polarity (<xref ref-type="bibr" rid="B2">Abraham and Leo, 1987</xref>). The core concept of HINT, as envisioned by Abraham, was that there was rich thermodynamic information encoded in LogP, and unlocking it would provide insight into more interaction phenomena than simple Newtonian physics-based molecular mechanics force fields. Kellogg joined Abraham at VCU in 1989 and fleshed out this concept by building a new, essentially <italic>de novo</italic>, modeling system that extended the connection matrix-based CLOG-P system of <xref ref-type="bibr" rid="B34">Hansch and Leo (1979)</xref> (and the Abraham and Leo enhancements) to the 3D world of structural biology. Thus, formulas for calculating atomistic LogPs were derived that retained the uniquely chemistry-aware features of the CLOG-P system, which itself relied on many thousands of careful measurements of organic and drug-like compounds. Additionally, methods for very quickly estimating or looking up solvent-accessible surface areas (SASAs) for atoms in small molecules, protein residues, and nucleotide bases were programmed (<xref ref-type="bibr" rid="B40">Kellogg and Abraham, 2000</xref>). Last, a &#x201c;distance function&#x201d; for representing the range effect of hydropathy was determined from some experimental observations indicating an exponential decay (<xref ref-type="bibr" rid="B36">Israelachvili and Pashley, 1982</xref>). The first and most obvious application of HINT was in calculating interaction scores between biomolecular species using the atomistic LogPs, atomistic SASAs, and this exponential decay distance function.</p>
<p>Inspired by the pioneering work of David Weininger at Daylight CIS and other researchers, the HINT program was rewritten in the late 1990s as a collection of object-oriented toolkit functions to enable the more facile creation of new application programs exploiting the HINT interaction model (<xref ref-type="bibr" rid="B42">Kellogg et al., 2005</xref>). The program is available as a set of toolkit functions by request from Glen Kellogg. The core functions were fully integrated into versions of the Sybyl program but do not, otherwise, have a turnkey application.</p>
<p>A major event in the history of HINT was the development of a very long-term collaboration between VCU&#x2019;s Abraham and Kellogg and Andrea Mozzarelli, Pietro Cozzini, and several exceptional students at the University of Parma. This collaboration and the numerous exciting results we found are reported in the following paragraphs. It should be noted that many of the innovations of the HINT model were inspired by the various interesting projects that the VCU and University of Parma teams carried out together.</p>
</sec>
<sec id="s2">
<title>2 HINT: definition and applications</title>
<sec id="s2-1">
<title>2.1 The HINT model</title>
<sec id="s2-1-1">
<title>2.1.1 Algorithms and code</title>
<p>The core of the HINT code is the assignment to each atom of a factor <italic>a</italic> that is derived from a LogP library coupled with algorithms to appropriately parse this information, where LogP is the partition function of an atom between water and 1-octanol. These two media were selected as representative of the polar and apolar environments within a protein. The atom-type LogP library was adapted from the CLOG-P approach (<xref ref-type="bibr" rid="B34">Hansch and Leo, 1979</xref>) with extensions (<xref ref-type="bibr" rid="B2">Abraham and Leo, 1987</xref>). HINT counts either positive or negative contributions for each individual atom&#x2013;atom interaction based on their hydropathic properties. By summing up these atomic contributions, an overall score is obtained. As LogP is a thermodynamic parameter, the HINT score is directly related to the free energy of complex formation. More explicitly, the interaction between two atoms, namely, <italic>i</italic> and <italic>j</italic>, is the product of their atom factors, called partial log P<sub>o/w</sub> (<italic>a</italic>
<sub>
<italic>i</italic>
</sub>) and solvent-accessible surface area (<italic>S</italic>
<sub>
<italic>i</italic>
</sub>):<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>f(r</italic>
<sub>
<italic>ij</italic>
</sub>
<italic>)</italic> represents a function of the distance between the two atoms, <italic>i</italic> and <italic>j</italic>. The atomistic <italic>S</italic>
<sub>
<italic>i</italic>
</sub> parameters are applied to represent &#x201c;exposure&#x201d; of that atom for interaction with atoms in other molecules. Atom&#x2013;atom distances are obtained either from the three-dimensional structure of the complex or from a model generated by docking procedures or homology modeling. The higher the resolution (or reliability) of the structures, the more precise the prediction (see <xref ref-type="sec" rid="s3-3">Section 3.3</xref>). Consequently, the total interaction <italic>B</italic> between two molecules is calculated by the (double) sum over all atom&#x2013;atom interactions:<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>A key feature of HINT is the exploitation of the experimentally determined LogP values, thus avoiding complex equations and approximations present in most of the other methods developed for predicting protein&#x2013;ligand interactions. The other conceptual advantage of using LogP is that this parameter provides an overall representation of the energetics of the encounter process between two molecules, without dividing the energetics into specific enthalpic contributions, such as electrostatic bonds, hydrogen bonds, and van der Waals bonds, a procedure that is not thermodynamically legitimate (<xref ref-type="bibr" rid="B24">Dill, 1997</xref>). In addition, and quite relevant, LogP implicitly includes the hydrophobic contribution generated by the change in th3e number of water molecules surrounding the interacting molecules before and upon the complex formation, i.e., the entropic contribution to the free energy (<xref ref-type="fig" rid="F1">Figures 1A, B</xref>). This contribution, which might be significant when apolar molecules interact, is usually not counted by most other programs or roughly approximated by calculating the accessible surface areas in contact between interacting molecules.</p>
</sec>
<sec id="s2-1-2">
<title>2.1.2 Why is HINT different?</title>
<p>The motivation behind HINT was to minimize the &#x201c;modeling&#x201d; and extract interaction information and guidance as much as possible from the experiment&#x2014;in this case, LogP. The measurement of LogP for a small molecule takes place in an environment that has gross similarities to that in which biology takes place, with one solvent (water) commingling with another (1-octanol) that is a stand-in for lipids and membranes. Nevertheless, a few tricks had to be applied, e.g., to make the Hansch and Leo <italic>fragment</italic> constants <italic>atomistic</italic> and to properly sign polar interactions (both Br&#xf8;nsted&#x2013;Lowry acids and bases are polar) (<xref ref-type="bibr" rid="B45">Kellogg et al., 1992</xref>; <xref ref-type="bibr" rid="B40">Kellogg and Abraham, 2000</xref>).</p>
<p>While quantitative aspects of programs like HINT are often used for comparative purposes, e.g., in dock scoring, virtual screening, and even LogP prediction, we have always tried to emphasize the qualitative value of HINT. Most importantly, HINT &#x201c;thinks&#x201d; like a medicinal chemist&#x2014;perhaps Donald Abraham in particular&#x2014;with respect to its language: <italic>hydrogen bonds</italic>, <italic>Lewis acids and bases</italic>, <italic>hydrophobic interactions</italic>, <italic>favorable</italic>, <italic>unfavorable</italic>, <italic>solvation</italic>, and <italic>desolvation.</italic> Thus, HINT&#x2019;s numerical or graphical output is always very interpretable, and it includes all biomolecular interactions to some degree, as they <italic>are</italic> all holistically encoded in LogP. HINT is, however, not necessarily as accurate as specialized software that focuses on specific phenomena, such as electrostatics, with finely tuned algorithms and parameters. It is, thus, difficult, if not impossible, to directly quantitate HINT&#x2019;s relative performance compared to other tools in a meaningful way: first, because there are no other tools in that space and, second, because &#x201c;improved understanding&#x201d; is not a definable metric.</p>
<p>However, to illustrate the quality of HINT code prediction for the strength of protein and nucleotide complexes, it was tested in many different types of interacting molecules and compared with experimentally determined dissociation constants. The overall results are remarkable, given the simplicity of the code and the speed of calculations. More importantly, virtually all of our studies produced results that suggested or proved new principles of biomolecular interaction. In the next sections, we summarize several cases where HINT has been applied.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<title>3 HINT applied to the evaluation of protein&#x2013;ligand interactions</title>
<p>Many proteins function via interactions with small ligands that exhibit a wide range of features covering most of the chemical space. For example, hemoglobin binds oxygen, which is a neutral molecule, intracellular hormone/vitamin receptors bind mostly apolar ligands, such as estrogens and retinoids, and proteases bind polar&#x2013;apolar polypeptide chains. This is made possible by sites where ligands bind that possess architectures dictated by the amino acids composing the pocket or site. Since members of the amino acid set are endowed with large differences in the polarity/apolarity of their sidechains, they can accommodate ionic and hydrophobic ligand interactions. In addition, the complementarity between the host (protein) and guest (ligand) may be obtained by optimized conformational changes of either one or the other of the interacting partners to obtain a &#x201c;perfect&#x201d; fit or, in a different view, by the selection of the protein and ligand conformations that perfectly match by steric and chemical complementarity. The net result is the formation of a binary complex characterized by an equilibrium between free and ligand-bound molecules regulated by a dissociation constant. The lower the value of the dissociation constant, the higher the match, i.e., the higher the number and overall strength of the interactions.</p>
<p>In order to evaluate how effectively HINT mimics the energetics of biological processes that involve the encounter of proteins and small ligands, we carried out a series of investigations exploring many of the variables that control the strength of protein&#x2013;ligand interaction. These variables are as follows: i) polarity/apolarity of ligands and/or protein active sites; ii) roles of ligand- and protein-bound water before and upon complex formation; iii) ionization state of ligand moieties and amino acid lateral chains contributing to shape the active sites, i.e., computational titration; iv) orientation of hydrogen atoms bound to polar residues and involved in H-bonds, i.e., the rank toolbox, and v) their actual contribution to the binding free energy, i.e., the relevance toolbox (<italic>vide infra</italic>) (<xref ref-type="fig" rid="F1">Figure 1C</xref>).</p>
<p>As structural information derived by X-ray crystallography is of paramount relevance to HINT analysis, as well as most of the codes aimed at the prediction of protein&#x2013;ligand energetics, the quality of the three-dimensional structures significantly impacts the confidence of the predicted scores.</p>
<sec id="s3-1">
<title>3.1 Free energy of interactions</title>
<p>We first evaluated by HINT the free energy of binding for 53 protein&#x2013;ligand complexes formed by 17 proteins of known three-dimensional structure and characterized by different active site polarities (<xref ref-type="bibr" rid="B54">Marabotti et al., 2000</xref>; <xref ref-type="bibr" rid="B22">Cozzini et al., 2002</xref>). This analysis was carried out without the contribution of bound water molecules to the binding energy. A successive analysis considered such contributions (see <xref ref-type="sec" rid="s4-1">Section 4.1</xref>) (<xref ref-type="bibr" rid="B21">Cozzini et al., 2004</xref>; <xref ref-type="bibr" rid="B29">Fornabaio et al., 2004</xref>; <xref ref-type="bibr" rid="B44">Kellogg et al., 2004</xref>). To demonstrate that HINT was able to correctly predict the interaction energy between protein and ligands, independently of ligand and protein active site polarity, the selected protein&#x2019;s active sites vary from very apolar, such as retinol-binding proteins, where the hydrophobic contribution and, consequently, the entropic contribution to the binding energy is predominant, to polar, such as penicillopepsin, where the free energy of interaction is mainly associated with contributions from Coulombic interaction between polar or ionizable residues. Another key feature of the analyzed set of protein&#x2013;ligand complexes is represented by the large range of binding strength that varies over nine orders of magnitude, with &#x394;G ranging from &#x2212;2 to &#x2212;15&#xa0;kcal/mol, as calculated by either inhibition constants or dissociation constants (<xref ref-type="bibr" rid="B22">Cozzini et al., 2002</xref>).</p>
<p>HINT scores for the protein&#x2013;ligand complexes were plotted against the experimental binding affinity, obtaining a remarkably good correlation with a standard error of 2.6&#xa0;kcal/mol, which translates into a prediction of affinity within about two orders of magnitude (<xref ref-type="fig" rid="F2">Figure 2</xref>). Even better standard errors (1&#xa0;kcal/mol) were obtained within single sets of specific ligand&#x2013;protein complexes, such as trypsin, thrombin, and tryptophan synthase. Given the polarity heterogeneity of the 53 protein&#x2013;ligand complexes, the prediction was very good.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Correlation between experimental &#x2206;G and HINT score units for 53 protein&#x2013;ligand complexes from more than 10 distinct proteins <bold>(A)</bold> characterized by a wide range of ligand polarity as indicated by their LogP values <bold>(B)</bold> (<xref ref-type="bibr" rid="B22">Cozzini et al., 2002</xref>). The line is the best least-squares fit.</p>
</caption>
<graphic xlink:href="fmolb-10-1194962-g002.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Effects of pH on interactions</title>
<p>A key requirement for obtaining good correlation between computational and experimental data, thus predicting binding affinity for ligands with known three-dimensional complex structures, is the homogeneity between the pH at which binding affinities are measured in solution and the pH at which structures have been determined. In fact, frequently, solution data are determined at pH values quite different from crystal structure medium. As a result, the protonation state of groups might be incorrectly attributed (or modeled), thus affecting the scores. A telling story is represented by an analysis of the binding affinity of penicillopepsin&#x2013;ligand complexes measured at pHs 3.5, 4.5, and 5.5 and by three-dimensional structures determined at the same pH values. HINT score prediction using these data produced a linear correlation with <italic>r</italic>
<sup>2</sup> equal to 0.99 (<xref ref-type="bibr" rid="B22">Cozzini et al., 2002</xref>), indicating that when pH values of solution and structure data match each other, the HINT prediction is remarkably good.</p>
<p>To address, in a more general way, the dependence of the HINT score on the protonation state of interacting groups of ligand and protein, a protocol called &#x201c;computational titration&#x201d; was designed (<xref ref-type="bibr" rid="B28">Fornabaio et al., 2003</xref>; <xref ref-type="bibr" rid="B44">Kellogg et al., 2004</xref>; <xref ref-type="bibr" rid="B43">2006</xref>; <xref ref-type="bibr" rid="B69">Spyrakis et al., 2004</xref>). The steps required by this method (<xref ref-type="fig" rid="F3">Figure 3</xref>) are as follows: i) the identification by a careful inspection of the three-dimensional structures of the ionizable groups of ligand and protein that can contribute to the free energy of binding; ii) the generation of all possible ionization state models for the interacting groups; iii) the energy minimization of the generated models with no movement of the heavy atoms (isocrystallographic models); and iv) the determination of the score for each model with HINT. This is a rather simple problem when only a few ionizable residues or functional groups are involved but increases rapidly in complexity. Each acid has three possible states (unprotonated and protonation at the two oxygens), while each amine or thiol has two, and&#x2014;if considered&#x2014;guanidyl could have five. The result is the dependence of the interaction free energy on group ionization states and the identification of the pattern of group ionization states, which leads to the optimal interaction strength. This, in turn, can be correlated with a well-defined pH value (<xref ref-type="bibr" rid="B69">Spyrakis et al., 2004</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Flowchart of the computational titration.</p>
</caption>
<graphic xlink:href="fmolb-10-1194962-g003.tif"/>
</fig>
<p>We applied the computational titration algorithm to the analysis of the interaction between neuraminidase and nine inhibitors for which three-dimensional structures and inhibition constants were known from the literature (<xref ref-type="bibr" rid="B28">Fornabaio et al., 2003</xref>), between HIV protease and its substrate peptide (<xref ref-type="bibr" rid="B69">Spyrakis et al., 2004</xref>) and between dihydrofolate reductase and three ligands (<xref ref-type="bibr" rid="B43">Kellogg et al., 2006</xref>). In the case of the computational titration procedure for neuraminidase, one proton at a time was introduced into the molecular model. Here, the protein possesses three ionizable amino acid sidechains, namely, Glu 119, Asp 151, and Glu 276, in the active site and one carboxylic moiety on one of the ligands. Thus, a total of four protons were added to the models, plus the starting 0-level model in which all ionizable groups are unprotonated. For each added proton, all possible models were generated, some of them being more chemically probable than others. As shown in <xref ref-type="bibr" rid="B28">Fornabaio et al. (2003)</xref>, the dependence of HINT score vs. added protons exhibits, not surprisingly, a bell shape, suggesting the optimal protonation state of the ionizable residues, and thus the pH for optimal interaction strength. A further step in the optimization of the computational titration, which requires building and energy minimizing sometimes thousands of models with specific protonation for each protein&#x2013;ligand complex, was to exclude high-energy models (chemically unplausible) and include a statistical mechanics evaluation of all models. This refined approach was applied to the computational titration of HIV protease and its peptidic substrate (<xref ref-type="bibr" rid="B69">Spyrakis et al., 2004</xref>). There are four ionizable residues in the active site, i.e., Asp 25&#x3b1;, 29&#x3b1;, 30&#x3b1;, and 25&#x3b2;, and four potentially interacting ionizable groups on the substrate, i.e., three carboxylates and one amine. In this case, 4,374 unique protonation models were generated, ranging from the most basic with all sites deprotonated (overall charge of &#x2212;7) to the most acidic with all sites protonated (overall charge of &#x2b;1). Titration analysis showed: i) a quite large range of HINT scores for models containing the same number of protons indicating favorable and unfavorable proton distributions (<xref ref-type="bibr" rid="B69">Spyrakis et al., 2004</xref>), ii) a sharp increase in the HINT score as protons were added that levels off upon five added protons, and iii) a good correlation between experimental data, collected between pH 5 and 3, and the Boltzmann-averaged HINT scores.</p>
</sec>
<sec id="s3-3">
<title>3.3 Resolution and prediction quality</title>
<p>Another factor that shows a profound impact on the predictions based on HINT, but also to all other codes, is the quality of the three-dimensional structures. The higher the resolution, the lower the standard error in the HINT prediction. We have observed that by selecting within the 53 complexes only structures determined with a resolution better than 2.5&#xa0;&#xc5;, the standard error of the prediction improved to 1.8&#xa0;kcal/mol compared to 2.6&#xa0;kcal/mol when the included complexes possess a resolution within 3.2&#xa0;&#xc5; (unpublished data). A better crystallographic resolution leads to a more precise geometry and well-defined distances of interacting groups that impact HINT scores.</p>
<p>Overall, the HINT score for a protein&#x2013;ligand complex depends, in addition to other contributions, on hydrogen-bonding contribution. A particular property of HINT is that it is very sensitive to the positioning and orientation of hydrogen-bonding protons. It is well known that hydrogen atoms can rarely be detected crystallographically, except for structures with resolutions higher than &#x223c;1&#xa0;&#xc5;. Thus, automated procedures were put in place to insert hydrogen atoms bound to heavy atoms and to correctly orient them to optimize the geometry for hydrogen bonds and, therefore, their strength and scores, without altering the position of the crystallographically determined heavy atoms. The energetic contribution of hydrogen bonds also plays a significant role when water molecules bound to protein active sites and/or ligands are considered (see the following section).</p>
</sec>
</sec>
<sec id="s4">
<title>4 HINT applied to the evaluation of protein&#x2013;ligand interactions: the contribution of water molecules</title>
<p>Nothing happens in biological processes without the direct or indirect contribution of water. Protein&#x2013;ligand complexes, as well as most biological complexes, form in water and often involve the displacement of many water molecules from protein binding sites. However, a few water molecules might be retained that further stabilize the complex association. Water displacement or retention participates in the overall free energy of binding, but the detailed estimation of each water molecule&#x2019;s contribution is often far from trivial. Therefore, hit identification and, most of all, lead optimization might suffer from the uncertainty of retaining or removing specific water molecules. This has been and continues to be an interesting problem to us, and we have attempted to rationalize water&#x2019;s role and contribution by means of the HINT code and specifically developed tools (<xref ref-type="fig" rid="F3">Figure 3</xref>), such as rank (<xref ref-type="bibr" rid="B41">Kellogg and Chen, 2004</xref>), to address hydrogen-bonding quality, free energy contributions of water&#x2013;ligand interactions (<xref ref-type="bibr" rid="B29">Fornabaio et al., 2004</xref>) and water&#x2013;protein interactions (<xref ref-type="bibr" rid="B11">Amadasi et al., 2006</xref>), the water relevance metric for assessing water conservation with respect to ligand binding (<xref ref-type="bibr" rid="B12">Amadasi et al., 2008</xref>), and a tool for similar to GRID (<xref ref-type="bibr" rid="B31">Goodford, 1985</xref>) for placing water molecules using relevance (<xref ref-type="bibr" rid="B42">Kellogg et al., 2005</xref>).</p>
<sec id="s4-1">
<title>4.1 HINT estimation of the energy contribution provided by bridging and conserved water molecules</title>
<p>The literature provides numerous cases in which water plays crucial roles in mediating protein&#x2013;ligand interactions. Examples are represented by bosutinib binding to Src kinase (<xref ref-type="bibr" rid="B52">Levinson and Boxer, 2014</xref>), thermolysin and its water network stabilized by carboxybenzyl-Gly-(PO<sub>2</sub>)-L-Leu-NH<sub>2</sub>&#x2212;based ligands (<xref ref-type="bibr" rid="B50">Krimmer et al., 2014</xref>), the water networks of the adenosine A2A receptor and their perturbations resulting from ligand binding (<xref ref-type="bibr" rid="B20">Congreve et al., 2012</xref>), or by the well-known case of HIV-1 protease (<xref ref-type="bibr" rid="B51">Lam et al., 1994</xref>). Trying to rationalize the associated water roles, we used HINT to estimate the energetics of binding site waters in 23 HIV-1 protease/inhibitor complexes that belong to the different classes of hydroxyethylenic ligands, peptidomimetic diol derivatives, cyclic ureidic derivatives, and cyclic sulfamide derivatives. The latter pair was specifically designed to displace a conserved water molecule (water 301) in the binding site. A careful analysis of the X-ray structures of the complexes showed the presence of three different structural water categories, that is, water 301, symmetrically located in the binding site with respect to the catalytic one and H-bonding Ile 50 and Ile 150 backbones, waters 313 and 313&#x2032;, located in a more peripheral region, and waters 313bis and 313bis&#x2019;, inserted in small clefts in the binding site. The protein&#x2013;ligand interaction calculated by HINT, with the inclusion of the water contribution, showed that bridging water molecules, such as 301, significantly contributes to the overall energy (about 4&#x2013;6&#xa0;kcal/mol) and leads to a better correlation between experimental and predicted free energy. In contrast, the contribution of more peripheral waters is variable or not significant and strongly dependent on the chemical nature and size of the ligand (<xref ref-type="bibr" rid="B29">Fornabaio et al., 2004</xref>). However, the binding energy gained by the new ligand-to-protein (Ile 50, Ile 150) interactions in the water 301-displacing cyclic derivatives was very similar to the energy previously associated with the water-to-ligand and water-to-protein interactions (<xref ref-type="fig" rid="F4">Figure 4</xref>). As a conclusion, we recommended including the contribution of water molecules when predicting the free energy of binding but paying specific attention to ensure that this adds information and not just noise.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Structural water molecules in the HIV-1 protease binding site. The residues lining the pocket (light blue), the ligand (pink), and the water molecules are shown in capped sticks. The protein is displayed in cartoons, and H-bonds involving water molecules are shown by gray dashed lines (PDB ID: 1HIH). The image has been obtained with PyMol version 2.x.</p>
</caption>
<graphic xlink:href="fmolb-10-1194962-g004.tif"/>
</fig>
<p>We implemented the previous work by evaluating not only the energy associated with protein&#x2013;water interaction but also the geometry quality of the H-bonds formed by each water molecule to obtain a more reliable estimation and prediction of water&#x2019;s role in mediating protein&#x2013;ligand association (<xref ref-type="bibr" rid="B11">Amadasi et al., 2006</xref>). First, 12 solvated proteins, in both the <italic>apo</italic> and <italic>holo</italic> form complexed with inhibitors, were considered for a total of 2,186 co-crystallized water molecules. Water molecules were classified and separated into the following categories: 1) water molecules in active sites; 2) water molecules deeply inserted into cavities; 3) buried water molecules; 4) first shell external water molecules (within 4&#xa0;&#xc5; of the nearest protein heavy atom); and 5) second shell external water molecules (more than 4&#xa0;&#xc5; from the protein; see the study of <xref ref-type="bibr" rid="B11">Amadasi et al. (2006)</xref>. Each water molecule was evaluated and classified according to the interactions made with the protein, by means of the HINT score, and to the number of potential H-bonds, by means of the rank algorithm (<xref ref-type="bibr" rid="B41">Kellogg and Chen, 2004</xref>).</p>
</sec>
<sec id="s4-2">
<title>4.2 The Rank algorithm</title>
<p>The latter can evaluate the count, strength, and geometry of potential hydrogen bonds for each water molecule in the cognate structure and is also used to optimize water position and hydrogen placement. Rank can vary from 0, for waters not involved in any hydrogen bonds, to around 6, for waters forming four high-quality hydrogen bonds in terms of bond length and angle geometry, and it is calculated by the following equation:<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mi>n</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2.80</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>&#xc5;</mml:mi>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mi mathvariant="italic">cos</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <italic>r</italic>
<sub>
<italic>n</italic>
</sub> is the distance between the water oxygen and the target heavy atom; <italic>&#x398;</italic>
<sub>
<italic>Td</italic>
</sub> is the ideal angle of 109.5&#xb0;, and <italic>&#x398;</italic>
<sub>
<italic>nm</italic>
</sub> is the angle between targets. Also, any angle less than 60&#xb0; is rejected, as well as the corresponding bond.</p>
<p>The analysis showed that both HINT and Rank scores increase when evaluating water molecules in the second and first hydration layers, to waters in active sites, up to waters in protein cavities, or completely buried in the protein matrix.</p>
<p>Then, 15 protein&#x2013;ligand complex binding sites, in which the presence of at least one bridging water molecule was reported in the literature, were analyzed to provide reference HINT scores and Rank values for bridging waters. We observed that the average Rank for protein&#x2013;water and ligand&#x2013;water was estimated to be 3.0 and 1.5, respectively, suggesting that proteins are better able to embed bridging waters than ligands, which is quite reasonable considering the residue sidechain flexibility and the presence of clefts and dips in active sites. We have to consider that a Rank equal to 3 might correspond to three H-bonds but also to two very good bonds in terms of distance and geometry. However, no significant difference was provided in terms of the HINT score for protein&#x2013;water and ligand&#x2013;water interactions.</p>
<p>Finally, we analyzed a set of nine proteins in both native and complexed state and classified water molecules in active sites in the following categories according to their relevance: 1) conserved water molecules in binding sites bridging protein&#x2013;ligand interaction; 2) conserved water molecules in binding sites not relevant for protein&#x2013;ligand interaction; 3) conserved water molecules in binding site cavities; 4) conserved water molecules in peripheral binding site regions; 5) water molecules displaced by ligands replacing their function, i.e., functionally displaced; 6) water molecules displaced by ligands only occupying their room, i.e., sterically displaced; and 7) missing waters. Water molecules with Rank &#x3e;1.5 and HINT score &#x3c;150 can be considered as sterically displaceable and, thus, potentially removable by ligands at moderate cost. Water molecules having, instead, Rank values in the 1.5&#x2013;4.0 range and HINT scores &#x3e;150 can be more relevant and quite likely retained with respect to drug design, while water molecules with Rank &#x3e;4 are too buried in the binding site and of less interest (<xref ref-type="bibr" rid="B11">Amadasi et al., 2006</xref>).</p>
</sec>
<sec id="s4-3">
<title>4.3 The Relevance metric</title>
<p>The diagnostic potential of the HINT score and Rank in predicting displaceable, bridging, or buried water molecules was further implemented in the more quantitative Relevance metric (<xref ref-type="bibr" rid="B12">Amadasi et al., 2008</xref>). Although water molecules are usually excluded from consideration in structure-based drug design, the rapid identification of Relevant water molecules that might be retained or displaced by <italic>ad hoc</italic> designed ligands (any water can be, in principle, displaced) can significantly affect the energetics of protein&#x2013;ligand interaction. By using a training set of thirteen proteins, crystallized in the native and complexed state, with a total number of 125 water molecules, we combined information coming from the HINT score and Rank in the following equation, describing the overall probability (<italic>P</italic>
<sub>
<italic>A</italic>
</sub>) of specific water to be Relevant:<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <italic>P</italic>
<sub>
<italic>R</italic>
</sub> and <italic>P</italic>
<sub>
<italic>H</italic>
</sub> are calculated by nonlinear polynomial regressions and correspond to the percent probability for conservation based on the Rank and HINT score, respectively, while <italic>W</italic>
<sub>
<italic>R</italic>
</sub> and <italic>W</italic>
<sub>
<italic>H</italic>
</sub> are the weights of Rank and HINT score probabilities, respectively (<xref ref-type="bibr" rid="B12">Amadasi et al., 2008</xref>). This analysis showed Rank and HINT score values associated with conserved waters: <italic>P</italic>
<sub>
<italic>R</italic>
</sub> &#x3e; 60%, corresponding to Rank &#x3e;2.3, and <italic>P</italic>
<sub>
<italic>H</italic>
</sub> &#x3e; 80%, corresponding to HINT score &#x3e;400. Otherwise, <italic>P</italic>
<sub>
<italic>H</italic>
</sub> &#x3e; 40%, although corresponding to HINT score &#x3e;100, was considered indicative of non-conservation. When applied on a test set of nine native and complexed proteins, for which specific bridging water molecules were previously identified, 59 of 68 water molecules were correctly predicted, corresponding to a success rate of 87%, which increased to 92% when only water molecules from X-ray structures having a resolution &#x3c;2.0&#xa0;&#xc5; were considered. Importantly, the crystallographic B factors, often used in such analyses did not improve this model.</p>
</sec>
<sec id="s4-4">
<title>4.4 Hot and cold water</title>
<p>In a later contribution, we designated water molecules in protein matrices as &#x201c;cold&#x201d; and &#x201c;hot&#x201d; according to their internal energy and possible displacement (<xref ref-type="bibr" rid="B65">Spyrakis et al., 2017</xref>). Hot water can also be considered &#x201c;unhappy&#x201d; water, that is, not stable in binding sites or at protein surfaces, because of the presence of extensive hydrophobic regions. Being unhappy, they would benefit from being displaced toward a more polar environment as the bulk, where they could maximize the number of H-bonds and provide a favorable variation of the binding free energy. Cold water, being stably bound in polar environments can be, instead, considered as &#x201c;happy&#x201d; molecules able to extend protein/ligand functions and to participate in the protein structure, dynamics, and function. Their displacement might not be at a trivial expense and may be associated with meaningless or even unfavorable variation of the binding free energy.</p>
<p>Behind the hot/cold classification is, in fact, their enthalpic and entropic contributions. Several considerations can be drawn that might be useful with respect to a drug design perspective (<xref ref-type="bibr" rid="B5">Ahmed et al., 2013</xref>; <xref ref-type="bibr" rid="B65">Spyrakis et al., 2017</xref>): i) hot water molecules in binding sites can be easily displaced with a manifest gain in entropy and a negligible loss in enthalpy; ii) attention must be paid when displacing cold water molecules since the entropic gain might not balance the enthalpic loss; iii) the enthalpy/entropy compensation associated with cold water displacement can lead to quite small changes in the binding free energy, thus often making drug design efforts ineffectual; iv) hot water networks around protein&#x2013;ligand complexes hide hydrophobic moieties stabilizing the association; and v) both hot and cold water molecules at the protein&#x2013;protein interface can provide precious indications when designing protein&#x2013;protein inhibitors.</p>
</sec>
<sec id="s4-5">
<title>4.5 Virtual screening with the HINT scoring function</title>
<p>Clearly, the emphasis on modern drug discovery has shifted over the past decade or so toward higher-throughput modeling approaches, such as virtual screening. Before using HINT for extensive docking experiments, we were curious whether docking scores from various scoring functions correlated better with RMSD (root-mean-squared distance) or free energy of binding. In other words, does reproducing a crystal structure by docking, which is how most scoring functions are optimized, necessarily produce accurate predictions of binding energy? We examined 19 protein&#x2013;ligand complexes for which X-ray crystallographic structures and binding energy data were available, and we calculated experimental vs. computationally-derived free energy correlation by means of the HINT free energy scoring tool and other scoring functions (<xref ref-type="bibr" rid="B66">Spyrakis et al., 2007a</xref>). Correlations drawn after re-docking the cognate ligands in their corresponding protein structures were generally better with the HINT scoring function (<xref ref-type="bibr" rid="B66">Spyrakis et al., 2007a</xref>). Also, it was obvious from this study that scoring functions uniquely calibrated for the dataset or sets under study should nearly always be preferable to universal scoring functions.</p>
<p>The HINT scoring function described earlier has indeed been a powerful tool for virtual screening (<xref ref-type="bibr" rid="B63">Salsi et al., 2010</xref>; <xref ref-type="bibr" rid="B25">Farzan et al., 2011</xref>; <xref ref-type="bibr" rid="B67">Spyrakis et al., 2014</xref>; <xref ref-type="bibr" rid="B58">Obaidullah et al., 2018</xref>; <xref ref-type="bibr" rid="B39">Kayastha et al., 2022</xref>) with a few caveats, including: 1) the scoring function is extremely sensitive to structure, especially properly optimized hydrogen-bonding interactions that include the donor hydrogen&#x2019;s position, and 2) while it is more than fast enough for scoring docking models, it is not, at present, in the &#x201c;giga&#x201d; docking timescale range. Thus, HINT, in our hands, has been used as a second scoring function after docking with programs like GOLD or AutoDock and their native scoring functions have been applied as primary filters. In this use case, our in-house studies generally have been relatively small-scale and speculative, i.e., purchasing 10&#x2013;100 compounds for actual assay with a variety of low-to-medium-throughput methods. Nevertheless, about one-in-five of the compounds identified by virtual screening with HINT scoring were active.</p>
<p>Three examples of this research are as follows: 1) 14 pentapeptides, binding to <italic>Haemophilus influenzae</italic> O-acetylserine sulfhydrylase (OASS) with affinities ranging from &#x3bc;M to mM, were identified with good correlation between computational and experimental data, excluding peptides bearing a positive charge, which are likely overestimated by HINT. These results, combined with X-ray structures of the three best complexes, defined a pharmacophoric scaffold for the design of peptidomimetic inhibitors for this enzyme (<xref ref-type="bibr" rid="B63">Salsi et al., 2010</xref>). 2) To identify new antimicrobials toward <italic>Treponema denticola</italic> cystalysin, we performed virtual screening on 9,357 compounds with the FLAPsite algorithm (<xref ref-type="bibr" rid="B14">Baroni et al., 2007</xref>), docked those most promising, and rescored with HINT in a consensus approach. Among the 17 compounds selected for testing, two showed IC<sub>50</sub>s in the low &#x3bc;M range, identifying interesting hits for the development of novel antimicrobials (<xref ref-type="bibr" rid="B67">Spyrakis et al., 2014</xref>). 3) The most recent publication using HINT in this way (<xref ref-type="bibr" rid="B39">Kayastha et al., 2022</xref>) details the discovery of three new potent eIF4A inhibitors that diminished the viability of diffuse large B-cell lymphoma (DLBCL) cells. The actives were discovered via target-based virtual screening in the MolPort database, followed by pharmacophore-based and analog screening. Modeling suggests that these compounds clamp eIF4A into mRNA in an ATP-independent manner and depress eIF4A-dependent oncogene expression. Eukaryotic translation initiation factor 4A is essential in translation initiation as it unwinds the secondary structure of messenger RNA upstream of the start codon and enables active downstream ribosomal recruitment.</p>
</sec>
</sec>
<sec id="s5">
<title>5 HINT applied to the evaluation of protein&#x2013;protein interactions</title>
<p>Protein&#x2013;protein interactions are emerging as one of the most important features of biological structures. Building an understanding of their contributions is challenging because of many of the aforementioned issues. However, an important advantage of the HINT model in evaluating protein&#x2013;protein interactions is that the key hydrophobic&#x2013;hydrophobic interactions are handled in a robust atom-to-atom manner (<xref ref-type="bibr" rid="B40">Kellogg and Abraham, 2000</xref>; <xref ref-type="bibr" rid="B64">Sarkar and Kellogg, 2010</xref>), not lumped together as lipophilic surface contact areas or other aggregate metrics. There are two types of protein&#x2013;protein interactions: the first between distinct separate proteins, which have many implications for biological function, metabolism, and diseases, and the second within a protein&#x2014;as in protein multimers, protein folding, and intramolecular associations. Although these definitions are broad and probably inclusive of most biology, we have applied HINT to several specific problems as we looked for insight into the link between structure and function.</p>
<sec id="s5-1">
<title>5.1 Dissecting the energetics of protein&#x2013;protein associations</title>
<p>While protein&#x2013;ligand interactions are complex on their own, the addition of more degrees of freedom when two proteins interact is another level of complexity. For example, in a typical protein&#x2013;ligand interaction, there will likely be only a few relevant water molecules, but at a protein&#x2013;protein interaction surface, there can be one or two dozen. Using the aforementioned computational titration algorithm (<xref ref-type="bibr" rid="B28">Fornabaio et al., 2003</xref>; <xref ref-type="bibr" rid="B69">Spyrakis et al., 2004</xref>; <xref ref-type="bibr" rid="B43">Kellogg et al., 2006</xref>) would typically involve 2&#x2013;3 ionizable residues or functional groups in a protein&#x2013;ligand complex, but again, at a protein&#x2013;protein surface, this could be a much larger number.</p>
<p>To explore the water issue, we adapted the rank and relevance algorithms to protein&#x2013;protein interactions in order to probe the roles of water molecules found specifically at protein&#x2013;protein interfaces (<xref ref-type="bibr" rid="B7">Ahmed et al., 2011</xref>; <xref ref-type="bibr" rid="B5">2013</xref>). The result was the surprising conclusion that less than one-quarter of waters found at such interfaces were engaged in truly structure-supporting bridging roles, and just over one-quarter of them had unfavorable interactions with respect to both proteins. The remainder, over one-half, had favorable interactions with one protein and unfavorable interactions with the other (<xref ref-type="bibr" rid="B7">Ahmed et al., 2011</xref>). We proposed a new structural motif&#x2014;the hydrophobic bubble&#x2014;to describe the phenomenon of one or a collection of water molecules &#x201c;trapped&#x201d; in a hydrophobic cavity formed by the protein&#x2013;protein association (<xref ref-type="fig" rid="F5">Figure 5</xref>). Moreover, we hypothesized that these bubbles are critical to protein function as points/regions of instability to assist in protein&#x2013;protein dissociation when needed. Two implications of these observations are worth mentioning: 1) it will be very difficult to predict the presence of water in such loci based only on energetics as they are by definition unstable, although water clusters do gain enthalpic advantage through their interactions with each other, and 2) these phenomena may present a unique opportunity for drug discovery at protein&#x2013;protein interfaces as they are the true &#x201c;hotspots&#x201d; (<xref ref-type="bibr" rid="B65">Spyrakis et al., 2017</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Cavity is formed during the association of the human placental RNase inhibitor (hRI) and human angiogenin (hAng) proteins to form the complex (pdbid: 1a4y). A closeup view of a portion of this inter-protein interface is shown. The cavity&#x2019;s extents are depicted by rendering in white dots; the green and purple contours represent the character of the proteins surrounding that cavity, hydrophobic and polar, respectively. What we are terming a &#x201c;hydrophobic bubble&#x201d; is found in the upper left region of the cavity as it encloses or traps three non-relevant waters in a largely hydrophobic (green) environment, where their strongest interactions may be amongst themselves. The other two waters in the cavity are in a polar (purple) region and are more relevant. See the study of <xref ref-type="bibr" rid="B7">Ahmed et al. (2011</xref>) for a further discussion on water molecules at protein&#x2013;protein interfaces. Note that displacement of the three non-relevant &#x201c;hot&#x201d; waters, likely as a cluster, may reveal a pocket large enough for targeting as a site for protein&#x2013;protein inhibition.</p>
</caption>
<graphic xlink:href="fmolb-10-1194962-g005.tif"/>
</fig>
<p>To further examine our insistence on the importance of water at protein&#x2013;protein interfaces, we reported a docking study on a small number of protein&#x2013;protein complexes both &#x201c;dry&#x201d;, as allowed by the native ZDOCK (<xref ref-type="bibr" rid="B61">Pierce and Weng, 2008</xref>; <xref ref-type="bibr" rid="B60">Pierce et al., 2011</xref>) program, and &#x201c;wet&#x201d;, where we tricked ZDOCK into recognizing interfacial water molecules (<xref ref-type="bibr" rid="B59">Parikh and Kellogg, 2014</xref>) (<xref ref-type="fig" rid="F6">Figure 6</xref>). The latter models were far superior by HINT score and numerous metrics developed by the Critical Assessment of PRedicted Interactions (CAPRI) communitywide experiment (<xref ref-type="bibr" rid="B37">Janin et al., 2003</xref>; <xref ref-type="bibr" rid="B57">Mendez et al., 2005</xref>), but creating such models was very tedious at that time due to the limitations inherent in the ZDOCK code, e.g., its automatic deletion of anything that was not in its library of 20&#xa0;amino acids. This, coupled with the inherent problems in predicting the presence of water molecules that are &#x201c;not&#x201d; energetically viable, although obviously present, set us in a different direction with respect to HINT-based structure prediction (<xref ref-type="sec" rid="s5-3">section 5.3</xref>), including protein&#x2013;protein docking.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>
<bold>(A)</bold> Unsolvated docking results for the HyHEL-63/HEL complex. The left panels overlay the predicted ligand poses (cyan) with the crystal structure (red), and the right panels illustrate the interactions of B/Tyr58 with ligand residues. This model is representative of 40% of generated poses found after clustering of the complete set of docking solutions but does not show native residue&#x2013;residue contacts. <bold>(B)</bold> Solvated docking results for the HyHEL-63/HEL complex. This model shows native water-mediated residue-residue contacts, with B/Tyr58 showing a water-mediated hydrogen-bonding network with C/Val99 and C/Asp101 (see the study of <xref ref-type="bibr" rid="B59">Parikh and Kellogg, 2014</xref>).</p>
</caption>
<graphic xlink:href="fmolb-10-1194962-g006.tif"/>
</fig>
</sec>
<sec id="s5-2">
<title>5.2 The intramolecular HINT score</title>
<p>As we were developing HINT, implementing an intramolecular score was a very simple addition. However, it was not immediately obvious how to apply it to interesting and important problems until we started thinking about the difficulties of dealing with low-resolution crystal structures and how their reflection data are collected and refined. When the resolution is high, the reflection data are more than adequate to model the structure within the electron density envelopes and features, such as hydrophobic interactions, are seen. However, at low resolution, somewhat crude molecular mechanics forcefield algorithms are applied, and hydrophobic interactions&#x2014;that are not explicit in these force fields&#x2014;are often lost. We used the intramolecular HINT score as an adjuvant to contemporary structure refinement protocols (<xref ref-type="bibr" rid="B48">Koparde et al., 2011</xref>). We showed that low-resolution structures refined by including our HINT intramolecular score-based protocol were significantly more native-like based on structure quality metrics.</p>
<p>More recently, <xref ref-type="bibr" rid="B3">Agosta et al. (2022)</xref> used the intramolecular HINT score to evaluate the stability of proteins in response to single (and multiple) site mutations. In particular, the SARS CoV-2 spike glycoprotein that exhibits interaction between its receptor-binding domain and the human angiotensin-converting enzyme 2, and has been seen to mutate rapidly with devastating consequences, was modeled and mutated <italic>in silico</italic>. The intramolecular HINT scores for the alpha, beta, gamma, delta, and omicron variants confirm that all mutated trimeric spike protein structures are similarly or more stable than the wild-type. In addition, these scores show that the receptor-binding domains of these mutants are more stable than the wild-type. The HINT intramolecular scoring is very rapid, and presuming the structures are well-modeled is suggestive that it may be useful as a pre-screening tool for as-yet undiscovered mutants for any protein.</p>
</sec>
<sec id="s5-3">
<title>5.3 Protein structure predictions and 3D hydropathic networks</title>
<p>The concept of interaction networks has been recognized for decades as it is a way to systematize protein structure, especially regarding hydrogen bonding. Our work with HINT has continually highlighted the parallel and often more important hydrophobic interactions. For example, it is not a coincidence that the &#x3b1;1&#x3b2;1 (and &#x3b1;2&#x3b2;2) interface of hemoglobin are largely unaffected by the deoxy to oxy hemoglobin transitions (and dominated by hydrophobic contacts between the two subunits), while the &#x3b1;1&#x3b2;2 interface, with just as many hydrogen bonds, has few hydrophobic interactions (<xref ref-type="bibr" rid="B1">Abraham et al., 1997</xref>) and shifts significantly in the transition. Also relevant are the &#x201c;unfavorable&#x201d; hydrophobic interactions (including those involving water), which are similarly key elements of protein structure. To systematize all of these observations and to create a platform for structure modeling, we developed a 3D interaction mapping paradigm wherein all interactions (polar and hydrophobic, favorable, and unfavorable) between a residue and its structural environment are visualized as contours with specific colors, strengths, and positions for each residue in a protein (<xref ref-type="bibr" rid="B6">Ahmed et al., 2015</xref>; <xref ref-type="bibr" rid="B4">Ahmed et al., 2019</xref>). To control backbone conformation, we adopted a chess square schema in which the Ramachandran plot is binned into 8 &#xd7; 8, 45&#xb0; &#xd7; 45&#xb0; chess squares (<xref ref-type="bibr" rid="B6">Ahmed et al., 2015</xref>). Further binning or parsing by sidechain conformations were applied as follows: <italic>&#x3c7;</italic>
<sub>1</sub> for asparagine, aspartic acid, cysteine, histidine, isoleucine, leucine, methionine, phenylalanine, proline, serine, threonine, tryptophan, and tyrosine, and <italic>&#x3c7;</italic>
<sub>1</sub> and <italic>&#x3c7;</italic>
<sub>2</sub> for arginine, glutamine, glutamic acid, and lysine. The former, thus, have three parses per chess square and the latter nine (<xref ref-type="bibr" rid="B9">AL Mughram et al., 2021</xref>). These maps can, interestingly, after binning, be clustered into limited sets of interaction motifs characteristic of residue type, solvent accessibility, and structural role. The number of clusters obtained is dependent on the population of the associated bin and the complexity of observed and anticipated interactions. For example, alanine sidechains show about four unique maps per chess square, while arginine can show as many as 18 maps in some of the nine <italic>&#x3c7;</italic>
<sub>1</sub>/<italic>&#x3c7;</italic>
<sub>2</sub> parses. Altogether, there are about 18,000 maps for our soluble proteins data set&#x2014;each represents a unique constellation of interacting atoms for that residue type, backbone conformation, and sidechain parse. For illustration, <xref ref-type="fig" rid="F7">Figure 7</xref> sets out example contoured maps for six diverse residue types, all in the &#x3b1;-helix region of the Ramachandran plot. Importantly, it is not the identity of those interacting atoms but their specific properties, e.g., hydrophobic, polar, hydrogen bond donor or acceptor, <italic>etc.</italic>, and water molecules can fill these roles just as other amino acid residues, cofactors or small molecules can. Because these roles are agnostic in terms of sequence, we call this paradigm &#x201c;3D interaction homology&#x201d; to emphasize that it is not the residue identity but its interactions that drive the structure.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Example contoured hydropathic interaction basis maps for six residue sidechain types. The full set for all residue types includes about 18,000 such maps. Each of these maps illustrate one observed collection of interactions&#x2014;discovered by 3D map clustering&#x2014;between the named residue and its environment, including all other residues and water. Each map is taken from the set calculated in the same alpha helix region of the Ramachandran plot, and all are contoured at largely similar iso-density levels. Two views are plotted for each case: left- the CA&#x2013;CB (z) axis is pointed up, and right- the CA&#x2013;CB axis is pointed out of the page. The interaction types are color-coded by type: green- favorable hydrophobic interactions, i.e., depicting hydrophobic interactions between the residue depicted and other atoms in its environment; purple- unfavorable hydrophobic (i.e., hydrophobic-polar) interactions; blue- favorable polar (e.g., hydrogen bonding) interactions; and red- unfavorable polar interactions. For more explanation, see the following: alanine- <xref ref-type="bibr" rid="B4">Ahmed et al. (2019</xref>), isoleucine- <xref ref-type="bibr" rid="B10">AL Mughram et al. (2023</xref>), serine and cysteine- <xref ref-type="bibr" rid="B18">Catalano et al. (2021</xref>), phenylalanine- <xref ref-type="bibr" rid="B10">AL Mughram et al. (2023</xref>), aspartic acid- <xref ref-type="bibr" rid="B35">Herrington and Kellogg (2021</xref>).</p>
</caption>
<graphic xlink:href="fmolb-10-1194962-g007.tif"/>
</fig>
<p>With this extensive set of in-hand data, &#x223c;18,000 maps abstracted from &#x223c;750,000 residues in our dataset, we have a complete, quantifiable picture of the hydropathic valence of each residue type and its potential roles in a protein&#x2019;s hydropathic interaction network. Exploiting this set of maps involves &#x201c;matching&#x201d; each backbone-aligned map&#x2019;s encoded interactions between residues. Because there are limited sets of such maps for residue type and backbone conformation, this is a comparatively simpler problem than <italic>de novo</italic> atomistic minimization. More importantly, however, using HINT scoring and its core interaction model has enabled our view of the structure to be built upon a free energy framework and to account for some more subtle features of interactions that are not necessarily detected by molecular mechanics-based approaches. For example, sidechain maps of aromatic residues, phenylalanine, tyrosine, and tryptophan, show evidence of pi&#x2013;pi stacking and pi&#x2013;cation interactions (<xref ref-type="bibr" rid="B9">AL Mughram et al., 2021</xref>). Also, the computational titration algorithm was applied to aspartic acid, glutamic acid, and histidine map generation and allowed pH tuning of those residues&#x2019; interaction profiles (<xref ref-type="bibr" rid="B35">Herrington and Kellogg, 2021</xref>). Likewise, an understanding of the large differences in structural role between the serine and cysteine was explored with our HINT/interaction maps approach (<xref ref-type="bibr" rid="B18">Catalano et al., 2021</xref>). While our first goal was to document the interactions in which these residues engage, we were also able to develop an improved understanding of their roles in protein structure: 1) serine is considered to be somewhat more polar than cysteine but is <italic>significantly</italic> more solvent exposed; 2) while both have similarly consistent interaction roles (&#x223c;50&#x2013;60% favorable polar, <xref ref-type="fig" rid="F8">Figure 8</xref>) regardless of their accessibility, clearly very few cysteines are found on the outside of proteins; 3) it is also interesting (<xref ref-type="fig" rid="F8">Figure 8</xref>) that bridging (-S&#x2013;S- bonded) cysteines exist in much more hydrophobic environments than their unbridged analogs, which may be mechanistically suggestive. In another more recent study, the aliphatic hydrophobic residues were examined in both soluble and membrane proteins (<xref ref-type="bibr" rid="B10">AL Mughram et al., 2023</xref>), which surprisingly revealed somewhat modest differences in interaction profiles for these residues in soluble proteins, where they are often buried, and membrane proteins, where they are exposed to lipids. The latter is particularly important because very few X-ray crystal or even cryo-EM structures of membrane proteins retain their native lipids (<xref ref-type="bibr" rid="B33">Guo, 2020</xref>).</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Residue interaction character as a function of solvent-accessible surface area. Each data point represents a cluster of interaction maps. The size of each marker is representative of the number of residues within that cluster (see legend). Left: the character of interactions made by serine residues are dominated (&#x223c;60%) by favorable polar (blue) with very minor contributions from hydrophobic interactions (green) that decrease from &#x223c;5% at low solvent accessibility to near zero at fully exposed. On average, serine&#x2019;s solvent exposure is around 50% (vertical line). Center: same graph for cysteine. The overall trends are quite similar, except that, on average, cysteine&#x2019;s solvent exposure is only &#x223c;10%, indicating that cysteine is far more likely to be found buried in a protein than on its surface. Right: same graph for S&#x2013;S bridged cysteine (cystine), where the average solvent exposure is now only about 7%. Also, the character of interactions made by cystine is dominated by unfavorable hydrophobic interactions (purple), followed by favorable hydrophobic. Thus, -S&#x2013;S- bridged cysteines are found most frequently in strongly hydrophobic environments that are buried. These data may provide insight into predictions of cysteine -S&#x2013;S- bridge formation in protein structures (see the study of <xref ref-type="bibr" rid="B18">Catalano et al., 2021</xref>).</p>
</caption>
<graphic xlink:href="fmolb-10-1194962-g008.tif"/>
</fig>
<p>In the last few years, AlphaFold2 and RoseTTAFold (<xref ref-type="bibr" rid="B13">Baek et al., 2021</xref>; <xref ref-type="bibr" rid="B38">Jumper et al., 2021</xref>) achieved an astonishing power of prediction of protein three-dimensional structures thanks to the application of deep learning methods that revolutionized the approach applied until then (<xref ref-type="bibr" rid="B32">Goulet and Cambillau, 2022</xref>) This new capacity has changed the landscape of structural biology and drug discovery in surprisingly fundamental ways. There remain many proteins poorly predicted by these tools, and the emerging information concerning membrane proteins has likely not been effectively incorporated in these algorithms, as those structures themselves are not well-understood, especially with respect to the role(s) of their native lipids. In some cases, the value of experimental structural characterization has even been questioned! It should be noted, however, that even if computational predictions generate accurately-folded protein models, they are not likely to produce models that have sidechain conformations accurate enough for structure-based drug discovery and design (Tong et al., 2021). Indeed, it is the subtle features of the structure that drive toward success in these endeavors. HINT and other modeling tools that focus on the subtle interactions beyond gross folding and salt bridges will continue to be relevant and useful.</p>
</sec>
</sec>
<sec id="s6">
<title>6 HINT applied to the evaluation of DNA&#x2013;ligand and DNA&#x2013;protein interactions</title>
<p>The inclusion in HINT of both hydrophilic and hydrophobic terms makes it able to predict energy contributions in DNA&#x2013;ligand and DNA&#x2013;protein complexes very well. Indeed, it is well known that the structure of DNA is stabilized by the stacking interactions between the planes of the nucleobases along the helix axis, which are the main factor in stabilizing the double helix (<xref ref-type="bibr" rid="B71">Yakovchuk et al., 2006</xref>), and by the hydrogen bonds between complementary base pairs. Stacking interactions are hydrophobic in nature, whereas hydrogen bonds are essentially hydrophilic (<xref ref-type="bibr" rid="B26">Feng et al., 2019</xref>); therefore, when other (macro)molecules bind to DNA, their interactions are conditioned by both contributions assessed by the HINT.</p>
<sec id="s6-1">
<title>6.1 Intercalation agents</title>
<p>The analysis of energetic interactions involving DNA by using HINT was exploited for the first time in the study of the effects of antineoplastic drugs, such as chlorambucil (an alkylating agent), and anthracycline antibiotics, such as doxorubicin, daunorubicin, and others (which act as intercalating agents). In the first example, HINT analysis of a computational model of the adducts derived by administration of chlorambucil to the shuttle vector plasmid pZ189 demonstrated that the methylenes, as well as the phenyl ring of this drug, can form favorable hydrophobic interactions with nucleotides near the adduct site in the minor groove of DNA, promoting alkylation at the N-3 position of adenine (<xref ref-type="bibr" rid="B70">Wang et al., 1994</xref>).</p>
<p>Next, HINT was used to study the selectivity of doxorubicin intercalation and binding in all 64 unique base pair quartet combinations. The results showed that the interactions between doxorubicin and the base pairs above and below the intercalation site are mainly polar and favorable, resulting from acid&#x2013;base interactions between the heteroatoms of the antineoplastic drug and the nucleotide bases. In addition, favorable hydrophobic contributions are also present. Specificity was mainly associated with hydrogen bonds, in particular with a base pair three positions away from the intercalating site (<xref ref-type="bibr" rid="B46">Kellogg et al., 1998</xref>). These analyses were subsequently extended to explore the binding of six different anthracycline antibiotics (doxorubicin, daunorubicin, hydroxydoxorubicin, 9-dehydroxydoxorubicin, adriamycinone, and daunomycinone) to 32 different DNA octamer sequences. The analysis showed that the differences in free energy among the various compounds were in line with that experimentally observed, indicating that HINT was able to pinpoint the energetic contributions of ligand functional groups most relevant to sequence specificity (<xref ref-type="bibr" rid="B17">Cashman et al., 2003</xref>). A further investigation of 65 doxorubicin analogs and their complexes with eight octamer DNA sequences allowed prediction of the net energetic contribution of several functional groups in the tested ligands and detection of selective ligands for the different combinations of DNA sequences simulated in this study (<xref ref-type="bibr" rid="B15">Cashman and Kellogg, 2004</xref>).</p>
<p>Analysis of the interactions of the two antibiotics gentamicin and paromomycin with 12 designed analogs with ribosomal RNA also showed that rings III and IV of these compounds are involved in important polar interactions with rRNA (<xref ref-type="bibr" rid="B16">Cashman et al., 2001</xref>). Therefore, HINT has proven to be a sensitive tool for dissecting interactions between nucleic acids and their ligands and for assessing the strength of interactions with existing drugs, as well as for predicting possible modifications of these molecules to improve their affinity and/or selectivity.</p>
</sec>
<sec id="s6-2">
<title>6.2 Interactions between proteins and DNA</title>
<p>HINT was tested for its ability to evaluate interactions of much more complex systems, such as protein&#x2013;DNA complexes. An initial study was conducted on the interactions between estrogen receptors (ER) alpha and beta and the corresponding estrogen-responsive elements (EREs) near estrogen-regulated genes (<xref ref-type="bibr" rid="B55">Marabotti et al., 2007</xref>). By analyzing the structure of the DNA-binding domain (DBD) of ER&#x3b1; and the homology-based model of ER&#x3b2; bound to the ERE sequence, it was possible to identify the residues that contributed most to the binding affinity. Furthermore, by mutating each nucleotide pair in the two halves of the ERE binding site with all other possible pairs, it was possible to understand how mutations in the different positions of ERE could affect the binding affinity in both complexes. The results showed that, consistent with the experimental results, ER&#x3b1; binds the consensus ERE sequence with higher affinity than ER&#x3b2; and that few amino acids and bases of the consensus sequences are involved in specific interactions. Specifically, HINT was able to discriminate with high sensitivity the affinity of ER&#x3b1;/ER&#x3b2; DBDs for ERE sequences, as well as for non-ERE sequences used as negative controls (glucocorticoid- and progestinic responsive elements), whereas DDNA (<xref ref-type="bibr" rid="B73">Zhou et al., 2005</xref>), another predictor of protein&#x2013;DNA interaction energies available at that date, did not. We hypothesized that the reasons for this failure was that the DDNA predictor was based on a knowledge-based statistical potential trained on a reference database not including protein&#x2013;DNA complexes. However, it is significant that the HINT code was not derived specifically from the analysis of DNA structures; therefore, this fact confirmed us the general validity of the hydropathic approach (<xref ref-type="bibr" rid="B55">Marabotti et al., 2007</xref>).</p>
<p>For this particular set of complexes, the specificity of protein&#x2013;DNA sequence binding did not appear to be much affected by water molecules. However, given the importance of the contribution of water in the thermodynamics of DNA&#x2013;protein recognition, this investigation was extended in parallel to 39 additional DNA&#x2013;protein complexes for which a three-dimensional structure was available (<xref ref-type="bibr" rid="B68">Spyrakis et al., 2007b</xref>). Thus, it could be shown that the inclusion of water molecules at the interface between protein and DNA (the so-called &#x201c;bridging waters&#x201d;) in the energetic contribution calculated by HINT improved the correlation between its score and the experimental free energy of association, with a lower standard error. The fraction of bridging waters in this set of experiments was only 3.5% of the water molecules detected in the 39 crystallographic complexes, consistent with the percentage of water mediating recognition between proteins and DNA identified previously (<xref ref-type="bibr" rid="B62">Reddy et al., 2001</xref>). It was also possible to observe that the orientation and binding strength of these water molecules depended more on the nature of the amino acid sidechain than on the type of DNA bases.</p>
</sec>
<sec id="s6-3">
<title>6.3 Protein&#x2013;DNA recognition</title>
<p>A more comprehensive study on energy-based prediction of the specific recognition between amino acid residues and nucleotide bases was subsequently performed on a dataset of 100 high-resolution protein&#x2013;DNA complexes (<xref ref-type="bibr" rid="B56">Marabotti et al., 2008</xref>) (<xref ref-type="fig" rid="F9">Figure 9</xref>) and used to predict specific contacts between amino acids and nucleotide bases in a set of 45 zinc finger&#x2013;DNA complexes identified by phage display selection (<xref ref-type="bibr" rid="B30">Ghosh et al., 2006</xref>). By applying the HINT code, three main questions could be answered: i) Which amino acid&#x2013;base pairs are energetically most relevant to achieve a specific interaction? ii) Are there energetic propensities that justify specific recognition of a nucleotide base by an amino acid? iii) Are bridging waters able to influence the specificity of amino acid&#x2013;base recognition? The results showed that the amino acids that interact most frequently with nucleotide bases are Arg, Asn, Lys, Gln, Thr, Ser, Asp, and Gly; in fact, HINT calculated that the sum of their contacts accounts for more than 70% of the total number of contacts. Arg-G, Asn-A, Asp-C, Gln-A, Glu-C, and Lys-G appear to be the most energetically favorable contacts (Arg-G being the interaction that accounts for about 2/5 of the total HINT score for the complexes), while Asn-T, Asp-G, Gln-T, Glu-G, Ile-T, Leu-T, Met-T, and Val-T are unfavorable interactions. The analyses also showed that the same amino acid&#x2013;nucleotide base pairs relevant to protein&#x2013;DNA interactions are also particularly involved in water-mediated interactions.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Complex between the wild-type gene-regulating protein ARC and the DNA (PDB ID: 1BDN). The four chains of the protein are represented with different shades of pink and with highlighted solvent-accessible surface area. In transparency, it is possible to see the secondary structure elements composing the protein. The color code for the nucleotides is as follows: A: red, T: blue, G: green, and C: yellow. Cyan balls represent water molecules. The image has been obtained with UCSF ChimeraX (version 1.5).</p>
</caption>
<graphic xlink:href="fmolb-10-1194962-g009.tif"/>
</fig>
<p>For some amino acid&#x2013;base pairs, it was also possible to calculate a &#x201c;water enhancement factor,&#x201d; that is, the ability of bridging waters to enhance the energetics of the amino acid&#x2013;base interaction (<xref ref-type="fig" rid="F10">Figure 10</xref>). Finally, based on the HINT score extracted from this analysis, it was possible to correctly predict more than 70% of the experimentally observed amino acid&#x2013;base pairs in the zinc finger&#x2013;DNA used as a test set. This percentage increased to nearly 90% when a relevance-weighted success descriptor, considering the relative energy relevance of each amino acid&#x2013;base pair to the total protein&#x2013;DNA recognition energy, was included. In this way, it was possible to show that amino acid&#x2013;nucleotide base preferences could be explained by the energy-based analysis performed by HINT better than through qualitative approaches based on purely geometric considerations. Moreover, HINT also made it possible to predict unfavorable interactions, some of which are surprisingly well-conserved and usually involve the methyl group of thymine. This finding is interesting in that one might speculate that the amino acid&#x2013;base interaction evolved before DNA development and was later adapted to DNA, but the presence of the thymine methyl group still continues to be a disruptive element in protein&#x2013;DNA interaction (<xref ref-type="bibr" rid="B56">Marabotti et al., 2008</xref>).</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Heat map describing the water enhancement factor (WEF), i.e., the HINT score enhancement due to water contribution, calculated for each amino acid&#x2013;base pair. A water enhancement factor of 1 indicates an amino acid (AA)&#x2013;base (B) interaction with no significant bridging water molecules. Data are extracted from <xref ref-type="bibr" rid="B56">Marabotti et al. (2008)</xref>.</p>
</caption>
<graphic xlink:href="fmolb-10-1194962-g010.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusion" id="s7">
<title>7 Conclusion</title>
<p>The prediction of events in antiquity was a very profitable but risky job, from the prophet Cassandra to those relying on Sibilla oracle cards. In addition to gambling, which, by definition, remains unpredictable, the forecast of weather is perhaps the most common modern-day testing ground for prediction. In use are algorithms that take into account the many variables that dictate sunny or rainy days, windy or calm weather conditions, and temperature. In this case, the robustness of the prediction can be and is verified every day, and the applied algorithms are constantly adjusted with incremental but beneficial improvements.</p>
<p>Prediction of the binding affinity between a protein and a ligand, or for any two biological molecules, independently of their molecular weights, is as challenging to predict as a protein structure. Multiplicity and diversity are the rules as many small energetic contributions lead to either loose, medium, or tight complexes. Some of these contributions are difficult to pinpoint as they deal with entropy or other emergent phenomena. HINT is one of the few codes that have attempted energetic evaluations of the molecular events associated with the formation of a protein&#x2013;ligand complex&#x2014;considering both enthalpic and entropic contributions&#x2014;in a very simple and &#x201c;natural&#x201d; way (<xref ref-type="bibr" rid="B40">Kellogg and Abraham, 2000</xref>). This idea was one of many innovative and transformative concepts that can be credited to Donald J. Abraham. He was an outstanding medicinal chemist who, by having visited Max Perutz&#x2019;s laboratory at the Cambridge MRC (Medical Research Council), had a clear appreciation of the value of protein structural information for the discovery, design, and development of novel drugs.</p>
<p>The results of applications of HINT to many diverse protein&#x2013;ligand and protein&#x2013;nucleotide complexes with and without water contributions, reported herein, demonstrate that it is possible to obtain very usable, if not accurate, predictions of protein&#x2013;ligand strength in short times and even with very low computational power. This paves the way for the design of chemical entities that correctly fit within protein active sites, enabling either inhibition or enhancement of their function, and potentially act as drugs to treat diseases. This was the dream of Abraham, and we are still pursuing it. We might be a bit closer!</p>
</sec>
</body>
<back>
<sec id="s8">
<title>Author contributions</title>
<p>GK, AMa, FS, and AMo planned, wrote, and discussed the manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s9">
<title>Funding</title>
<p>This work was supported by the University of Salerno (grant numbers ORSA199808, ORSA208455, and ORSA219407); MIUR (grant FFABR2017 and PRIN 2017 program, grant number 2017483NH8); BANCA D&#x2019;ITALIA (AMa); and the University of Turin (Ricerca Locale 2020, 2021) SPY_RILO_20_01, SPY_RILO_21_01 (FS).</p>
</sec>
<ack>
<p>The authors are deeply indebted to colleagues and students who contributed to the studies reported in the present review.</p>
</ack>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Holt</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Ackers</surname>
<given-names>G. K.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Hydropathic analysis of the non-covalent interactions between molecular subunits of structurally characterized hemoglobins</article-title>. <source>J. Mol. Bio.</source> <volume>272</volume>, <fpage>613</fpage>&#x2013;<lpage>632</lpage>. <pub-id pub-id-type="doi">10.1006/jmbi.1997.1249</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Leo</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>1987</year>). <article-title>Extension of the fragment method to calculate amino acid zwitterion and side chain partition coefficients</article-title>. <source>Proteins</source> <volume>2</volume>, <fpage>130</fpage>&#x2013;<lpage>152</lpage>. <pub-id pub-id-type="doi">10.1002/prot.340020207</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Agosta</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>From oncoproteins to spike proteins: The evaluation of intramolecular stability using hydropathic force field</article-title>. <source>J. Comput. Aided Mol. Des.</source> <volume>36</volume>, <fpage>797</fpage>&#x2013;<lpage>804</lpage>. <pub-id pub-id-type="doi">10.1007/s10822-022-00477-y</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Catalano</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Portillo</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Safo</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Scarsdale</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>3D interaction homology: The hydropathic interaction environments of even alanine are diverse and provide novel structural insight</article-title>. <source>J. Struct. Biol.</source> <volume>207</volume>, <fpage>183</fpage>&#x2013;<lpage>198</lpage>. <pub-id pub-id-type="doi">10.1016/j.jsb.2019.05.007</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Habtemariam</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Safo</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Scarsdale</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Unintended consequences? Water molecules at biological and crystallographic protein-protein interfaces</article-title>. <source>Comput. Biol. Chem.</source> <volume>47</volume>, <fpage>126</fpage>&#x2013;<lpage>141</lpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2013.08.009</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Koparde</surname>
<given-names>V. N.</given-names>
</name>
<name>
<surname>Safo</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Scarsdale</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>3D interaction homology: The structurally known rotamers of tyrosine derive from a surprisingly limited set of information-rich hydropathic interaction environments described by maps</article-title>. <source>Proteins</source> <volume>83</volume>, <fpage>1118</fpage>&#x2013;<lpage>1136</lpage>. <pub-id pub-id-type="doi">10.1002/prot.24813</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Tripathi</surname>
<given-names>P. K.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Scarsdale</surname>
<given-names>J. N.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Bound water at protein-protein interfaces: Partners, roles and hydrophobic bubbles as a conserved motif</article-title>. <source>PLoS One</source> <volume>6</volume>, <fpage>e24712</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0024712</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ajay</surname>
</name>
<name>
<surname>Murcko</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Computational methods to predict binding free energy in ligand-receptor complexes</article-title>. <source>J. Med. Chem.</source> <volume>38</volume>, <fpage>4953</fpage>&#x2013;<lpage>4967</lpage>. <pub-id pub-id-type="doi">10.1021/jm00026a001</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al Mughram</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Catalano</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bowry</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Safo</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Scarsdale</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>3D interaction homology: Hydropathic Analyses of the &#x201c;&#x3c0;-cation&#x201d; and &#x201c;&#x3c0;-&#x3c0;&#x201d; interaction motifs in phenylalanine, tyrosine, and tryptophan residues</article-title>. <source>J. Chem. Inf. Model.</source> <volume>61</volume>, <fpage>2937</fpage>&#x2013;<lpage>2956</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.1c00235</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al Mughram</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Catalano</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Herrington</surname>
<given-names>N. B.</given-names>
</name>
<name>
<surname>Safo</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>3D interaction homology: The hydrophobic residues alanine, isoleucine, leucine, proline and valine play different structural roles in soluble and membrane proteins</article-title>. <source>Front. Mol. Biosci.</source> <volume>10</volume>, <fpage>1116868</fpage>. <pub-id pub-id-type="doi">10.3389/fmolb.2023.1116868</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amadasi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Mapping the energetics of water-protein and water-ligand interactions with the "natural" HINT forcefield: Predictive tools for characterizing the roles of water in biomolecules</article-title>. <source>J. Mol. Biol.</source> <volume>358</volume>, <fpage>289</fpage>&#x2013;<lpage>309</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2006.01.053</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amadasi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Surface</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Robust classification of "relevant" water molecules in putative protein binding sites</article-title>. <source>J. Med. Chem.</source> <volume>51</volume>, <fpage>1063</fpage>&#x2013;<lpage>1067</lpage>. <pub-id pub-id-type="doi">10.1021/jm701023h</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baek</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>DiMaio</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Anishchenko</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Dauparas</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ovchinnikov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>G. R.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Accurate prediction of protein structures and interactions using a three-track neural network</article-title>. <source>Science</source> <volume>373</volume>, <fpage>871</fpage>&#x2013;<lpage>876</lpage>. <pub-id pub-id-type="doi">10.1126/science.abj8754</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baroni</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cruciani</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sciabola</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Perruccio</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Mason</surname>
<given-names>J. S.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>A common reference framework for analyzing/comparing proteins and ligands. Fingerprints for ligands and proteins (FLAP): Theory and application</article-title>. <source>J. Chem. Inf. Model.</source> <volume>47</volume>, <fpage>279</fpage>&#x2013;<lpage>294</lpage>. <pub-id pub-id-type="doi">10.1021/ci600253e</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cashman</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>A computational model for anthracycline binding to DNA: Tuning groove-binding intercalators for specific sequences</article-title>. <source>J. Med. Chem.</source> <volume>47</volume>, <fpage>1360</fpage>&#x2013;<lpage>1374</lpage>. <pub-id pub-id-type="doi">10.1021/jm030529h</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cashman</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Rife</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Which aminoglycoside ring is most important for binding? A hydropathic analysis of gentamicin, paromomycin, and analogues</article-title>. <source>Bioorg. Med. Chem. Lett.</source> <volume>11</volume>, <fpage>119</fpage>&#x2013;<lpage>122</lpage>. <pub-id pub-id-type="doi">10.1016/s0960-894x(00)00615-6</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cashman</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Scarsdale</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Hydropathic analysis of the free energy differences in anthracycline antibiotic binding to DNA</article-title>. <source>Nucleic Acids Res.</source> <volume>31</volume>, <fpage>4410</fpage>&#x2013;<lpage>4416</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkg645</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Catalano</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Al Mughram</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>3D interaction homology: Hydropathic interaction environments of serine and cysteine are strikingly different and their roles adapt in membrane proteins</article-title>. <source>Curr. Res. Struct. Biol.</source> <volume>3</volume>, <fpage>239</fpage>&#x2013;<lpage>256</lpage>. <pub-id pub-id-type="doi">10.1016/j.crstbi.2021.09.002</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cavasotto</surname>
<given-names>C. N.</given-names>
</name>
<name>
<surname>Aucar</surname>
<given-names>M. G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>High-throughput docking using quantum mechanical scoring</article-title>. <source>Front. Chem.</source> <volume>8</volume>, <fpage>00246</fpage>. <pub-id pub-id-type="doi">10.3389/fchem.2020.00246</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Congreve</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Andrews</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Dore</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Hollenstein</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hurrell</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Langmead</surname>
<given-names>C. J.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Discovery of 1,2,4-triazine derivatives as adenosine A(2A) antagonists using structure based drug design</article-title>. <source>J. Med. Chem.</source> <volume>55</volume>, <fpage>1898</fpage>&#x2013;<lpage>1903</lpage>. <pub-id pub-id-type="doi">10.1021/jm201376w</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fornabaio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Marabotti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Free energy of ligand binding to proteins: Evaluation of the contribution of water molecules by computational methods</article-title>. <source>Curr. Med. Chem.</source> <volume>11</volume>, <fpage>1345</fpage>&#x2013;<lpage>1359</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fornabaio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Marabotti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Simple, intuitive calculations of free energy of binding for protein-ligand complexes. 1. Models without explicit constrained water</article-title>. <source>J. Med. Chem.</source> <volume>45</volume>, <fpage>2469</fpage>&#x2013;<lpage>2483</lpage>. <pub-id pub-id-type="doi">10.1021/jm0200299</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Costantino</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Emerson</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Target flexibility: An emerging consideration in drug discovery and design</article-title>. <source>J. Med. Chem.</source> <volume>51</volume>, <fpage>6237</fpage>&#x2013;<lpage>6255</lpage>. <pub-id pub-id-type="doi">10.1021/jm800562d</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dill</surname>
<given-names>K. A.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Additivity principles in biochemistry</article-title>. <source>J. Biol. Chem.</source> <volume>272</volume>, <fpage>701</fpage>&#x2013;<lpage>704</lpage>. <pub-id pub-id-type="doi">10.1074/jbc.272.2.701</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Farzan</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Palermo</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Yokoyama</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Orefice</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Fornabaio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sarkar</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Premature activation of the paramyxovirus fusion protein before target cell attachment with corruption of the viral fusion machinery</article-title>. <source>J. Biol. Chem.</source> <volume>286</volume> (<issue>44</issue>), <fpage>37945</fpage>&#x2013;<lpage>37954</lpage>. <pub-id pub-id-type="doi">10.1074/jbc.M111.256248</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sosa</surname>
<given-names>R. P.</given-names>
</name>
<name>
<surname>M&#xe5;rtensson</surname>
<given-names>A. K. F.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dorfman</surname>
<given-names>K. D.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Hydrophobic catalysis and a potential biological role of DNA unstacking induced by environment effects</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>116</volume>, <fpage>17169</fpage>&#x2013;<lpage>17174</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1909122116</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Foloppe</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hubbard</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Towards predictive ligand design with free-energy based computational methods?</article-title> <source>Curr. Med. Chem.</source> <volume>13</volume>, <fpage>3583</fpage>&#x2013;<lpage>3608</lpage>. <pub-id pub-id-type="doi">10.2174/092986706779026165</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fornabaio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Simple, intuitive calculations of free energy of binding for protein-ligand complexes. 2. Computational titration and pH effects in molecular models of neuraminidase-inhibitor complexes</article-title>. <source>J. Med. Chem.</source> <volume>46</volume>, <fpage>4487</fpage>&#x2013;<lpage>4500</lpage>. <pub-id pub-id-type="doi">10.1021/jm0302593</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fornabaio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Simple, intuitive calculations of free energy of binding for protein-ligand complexes. 3. The free energy contribution of structural water molecules in HIV-1 protease complexes</article-title>. <source>J. Med. Chem.</source> <volume>47</volume>, <fpage>4507</fpage>&#x2013;<lpage>4516</lpage>. <pub-id pub-id-type="doi">10.1021/jm030596b</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghosh</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Stains</surname>
<given-names>C. I.</given-names>
</name>
<name>
<surname>Ooi</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Segal</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Direct detection of double-stranded DNA: Molecular methods and applications for DNA diagnostics</article-title>. <source>Mol. Biosyst.</source> <volume>2</volume>, <fpage>551</fpage>&#x2013;<lpage>560</lpage>. <pub-id pub-id-type="doi">10.1039/b611169f</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goodford</surname>
<given-names>P. J.</given-names>
</name>
</person-group> (<year>1985</year>). <article-title>A computational procedure for determining energetically favorable binding sites on biologically important macromolecules</article-title>. <source>J. Med. Chem.</source> <volume>28</volume>, <fpage>849</fpage>&#x2013;<lpage>857</lpage>. <pub-id pub-id-type="doi">10.1021/jm00145a002</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goulet</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cambillau</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Present impact of AlphaFold2 revolution on structural biology, and an illustration with the structure prediction of the bacteriophage J-1 host adhesion device</article-title>. <source>Front. Mol. Biosci.</source> <volume>9</volume>, <fpage>907452</fpage>. <pub-id pub-id-type="doi">10.3389/fmolb.2022.907452</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>Y. Z.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Be cautious with crystal structures of membrane proteins or complexes prepared in detergents</article-title>. <source>Crystals</source> <volume>10</volume>, <fpage>86</fpage>. <pub-id pub-id-type="doi">10.3390/cryst10020086</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hansch</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Leo</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>1979</year>). <source>Substituent constants for correlation analysis in chemistry and biology</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Wiley</publisher-name>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Herrington</surname>
<given-names>N. B.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>3D interaction homology: Computational titration of aspartic acid, glutamic acid and histidine can create pH-tunable hydropathic environment maps</article-title>. <source>Front. Mol. Biosci.</source> <volume>8</volume>, <fpage>773385</fpage>. <pub-id pub-id-type="doi">10.3389/fmolb.2021.773385</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Israelachvili</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pashley</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1982</year>). <article-title>The hydrophobic interaction is long range, decaying exponentially with distance</article-title>. <source>Nature</source> <volume>300</volume>, <fpage>341</fpage>&#x2013;<lpage>342</lpage>. <pub-id pub-id-type="doi">10.1038/300341a0</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Janin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Henrick</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Moult</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Eyck</surname>
<given-names>L. T.</given-names>
</name>
<name>
<surname>Sternberg</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Vajda</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2003</year>). <article-title>Capri: A critical assessment of predicted interactions</article-title>. <source>Proteins</source> <volume>52</volume>, <fpage>2</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1002/prot.10381</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jumper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Figurnov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ronneberger</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Highly accurate protein structure prediction with AlphaFold</article-title>. <source>Nature</source> <volume>596</volume>, <fpage>583</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kayastha</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Herrington</surname>
<given-names>N. B.</given-names>
</name>
<name>
<surname>Kapadia</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Roychowdhury</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nanaji</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Novel eIF4A1 inhibitors with anti-tumor activity in lymphoma</article-title>. <source>Mol. Med.</source> <volume>28</volume>, <fpage>101</fpage>. <pub-id pub-id-type="doi">10.1186/s10020-022-00534-0</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Hydrophobicity: Is LogP(o/w) more than the sum of its parts?</article-title> <source>Eur. J. Med. Chem.</source> <volume>35</volume>, <fpage>651</fpage>&#x2013;<lpage>661</lpage>. <pub-id pub-id-type="doi">10.1016/s0223-5234(00)00167-7</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>D. L.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>The importance of being exhaustive. Optimization of bridging structural water molecules and water networks in models of biological systems</article-title>. <source>Chem. Biodivers.</source> <volume>1</volume>, <fpage>98</fpage>&#x2013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1002/cbdv.200490016</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Fornabaio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>New application design for a 3D hydropathic map&#x2013;based search for potential water molecules bridging between protein and ligand, internet electron</article-title>. <source>J. Mol. Des.</source> <volume>4</volume>, <fpage>194</fpage>&#x2013;<lpage>209</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Fornabaio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>Tools for building a comprehensive modeling system for virtual screening under real biological conditions: The Computational Titration algorithm</article-title>. <source>J. Mol. Graph. Model.</source> <volume>24</volume>, <fpage>434</fpage>&#x2013;<lpage>439</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmgm.2005.09.001</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Fornabaio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Lodola</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2004</year>). <article-title>Getting it right. Modeling of pH, solvent and "nearly" everything else in virtual screening of biological targets</article-title>. <source>J. Mol. Graph. Model.</source> <volume>22</volume>, <fpage>479</fpage>&#x2013;<lpage>486</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmgm.2004.03.008</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Joshi</surname>
<given-names>G. S.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>1992</year>). <article-title>New tools for modeling and understanding hydrophobicity and hydrophobic interactions</article-title>. <source>Med. Chem. Res.</source> <volume>1</volume>, <fpage>444</fpage>&#x2013;<lpage>453</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Scarsdale</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Fornari</surname>
<given-names>F. A.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Identification and hydropathic characterization of structural features affecting sequence specificity for doxorubicin intercalation into DNA double-stranded polynucleotides</article-title>. <source>Nucleic Acids Res.</source> <volume>26</volume>, <fpage>4721</fpage>&#x2013;<lpage>4732</lpage>. <pub-id pub-id-type="doi">10.1093/nar/26.20.4721</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kirsopp</surname>
<given-names>J. J. M.</given-names>
</name>
<name>
<surname>Di Paola</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Manrique</surname>
<given-names>D. Z.</given-names>
</name>
<name>
<surname>Krompiec</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Greene-Diniz1</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Guba</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Quantum computational quantification of protein-ligand interactions</article-title>. <comment>arXiv, 2110.08163v1</comment>.</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koparde</surname>
<given-names>V. N.</given-names>
</name>
<name>
<surname>Scarsdale</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Applying an empirical hydropathic forcefield in refinement may improve low-resolution protein X-ray crystal structures</article-title>. <source>PLoS One</source> <volume>6</volume>, <fpage>e15920</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0015920</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koshland</surname>
<given-names>D. E.</given-names>
</name>
</person-group> (<year>1958</year>). <article-title>Application of a theory of enzyme specificity to protein synthesis</article-title>. <source>Proc. Natl. Acad. Sci. USA.</source> <volume>44</volume>, <fpage>98</fpage>&#x2013;<lpage>104</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.44.2.98</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krimmer</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Betz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Heine</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Klebe</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Methyl, ethyl, propyl, butyl: Futile but not for water, as the correlation of structure and thermodynamic signature shows in a congeneric series of thermolysin inhibitors</article-title>. <source>Chem. Med. Chem.</source> <volume>9</volume>, <fpage>833</fpage>&#x2013;<lpage>846</lpage>. <pub-id pub-id-type="doi">10.1002/cmdc.201400013</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lam</surname>
<given-names>P. Y.</given-names>
</name>
<name>
<surname>Jadhav</surname>
<given-names>P. K.</given-names>
</name>
<name>
<surname>Eyermann</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Hodge</surname>
<given-names>C. N.</given-names>
</name>
<name>
<surname>Ru</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bacheler</surname>
<given-names>L. T.</given-names>
</name>
<etal/>
</person-group> (<year>1994</year>). <article-title>Rational design of potent, bioavailable, nonpeptide cyclic ureas as HIV protease inhibitors</article-title>. <source>Science</source> <volume>263</volume>, <fpage>380</fpage>&#x2013;<lpage>384</lpage>. <pub-id pub-id-type="doi">10.1126/science.8278812</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Levinson</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Boxer</surname>
<given-names>S. G.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>A conserved water-mediated hydrogen bond network defines bosutinib&#x27;s kinase selectivity</article-title>. <source>Nat. Chem. Biol.</source> <volume>10</volume>, <fpage>127</fpage>&#x2013;<lpage>132</lpage>. <pub-id pub-id-type="doi">10.1038/nchembio.1404</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tsai</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Nussinov</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Folding funnels and binding mechanisms</article-title>. <source>Protein Eng.</source> <volume>12</volume>, <fpage>713</fpage>&#x2013;<lpage>720</lpage>. <pub-id pub-id-type="doi">10.1093/protein/12.9.713</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marabotti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Balestreri</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>HINT predictive analysis of binding between Retinol Binding Protein and hydrophobic ligands</article-title>. <source>Bioorg. Med. Chem. Lett.</source> <volume>10</volume>, <fpage>2129</fpage>&#x2013;<lpage>2132</lpage>. <pub-id pub-id-type="doi">10.1016/s0960-894x(00)00414-5</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marabotti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Colonna</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Facchiano</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>New computational strategy to analyze the interactions of ERalpha and ERbeta with different ERE sequences</article-title>. <source>J. Comput. Chem.</source> <volume>28</volume>, <fpage>1031</fpage>&#x2013;<lpage>1041</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.20582</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marabotti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Facchiano</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Alberti</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Energy-based prediction of amino acid-nucleotide base recognition</article-title>. <source>J. Comput. Chem.</source> <volume>29</volume>, <fpage>1955</fpage>&#x2013;<lpage>1969</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.20954</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mendez</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Leplae</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lensink</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Wodak</surname>
<given-names>S. J.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Assessment of CAPRI predictions in rounds 3&#x2013;5 shows progress in docking procedures</article-title>. <source>Proteins</source> <volume>60</volume>, <fpage>150</fpage>&#x2013;<lpage>169</lpage>. <pub-id pub-id-type="doi">10.1002/prot.20551</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Obaidullah</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Kitten</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Inhibiting pneumococcal surface antigen A (PsaA) with small molecules discovered through virtual screening: Steps toward validating a potential target for Streptococcus pneumoniae</article-title>. <source>Chem. Biodivers.</source> <volume>15</volume>, <fpage>e1800234</fpage>. <pub-id pub-id-type="doi">10.1002/cbdv.201800234</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parikh</surname>
<given-names>H. I.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Intuitive, but not simple: Including explicit water molecules in protein-protein docking simulations improves model quality</article-title>. <source>Proteins</source> <volume>82</volume>, <fpage>916</fpage>&#x2013;<lpage>932</lpage>. <pub-id pub-id-type="doi">10.1002/prot.24466</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pierce</surname>
<given-names>B. G.</given-names>
</name>
<name>
<surname>Hourai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Accelerating protein docking in ZDOCK using an advanced 3D convolution library</article-title>. <source>PLoS One</source> <volume>6</volume>, <fpage>e24657</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0024657</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pierce</surname>
<given-names>B. G.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>A combination of rescoring and refinement significantly improves protein docking performance</article-title>. <source>Proteins</source> <volume>72</volume>, <fpage>270</fpage>&#x2013;<lpage>279</lpage>. <pub-id pub-id-type="doi">10.1002/prot.21920</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reddy</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jayaram</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Do water molecules mediate protein-DNA recognition?</article-title> <source>J. Mol. Biol.</source> <volume>314</volume>, <fpage>619</fpage>&#x2013;<lpage>632</lpage>. <pub-id pub-id-type="doi">10.1006/jmbi.2001.5154</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salsi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Bayden</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Amadasi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Campanini</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bettati</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Design of O-acetylserine sulfhydrylase inhibitors by mimicking nature</article-title>. <source>J. Med. Chem.</source> <volume>53</volume>, <fpage>345</fpage>&#x2013;<lpage>356</lpage>. <pub-id pub-id-type="doi">10.1021/jm901325e</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sarkar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Hydrophobicity--shake flasks, protein folding and drug discovery</article-title>. <source>Curr. Top. Med. Chem.</source> <volume>10</volume>, <fpage>67</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.2174/156802610790232233</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Bayden</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>The roles of water in the protein matrix: A largely untapped resource for drug discovery</article-title>. <source>J. Med. Chem.</source> <volume>60</volume>, <fpage>6781</fpage>&#x2013;<lpage>6827</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jmedchem.7b00057</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Amadasi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fornabaio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<etal/>
</person-group> (<year>2007a</year>). <article-title>The consequences of scoring docked ligand conformations using free energy correlations</article-title>. <source>Eur. J. Med. Chem.</source> <volume>42</volume>, <fpage>921</fpage>&#x2013;<lpage>933</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejmech.2006.12.037</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cellini</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bruno</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Benedetti</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Carosati</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Cruciani</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Targeting cystalysin, a virulence factor of Treponema denticola-supported periodontitis</article-title>. <source>ChemMedChem</source> <volume>9</volume>, <fpage>1501</fpage>&#x2013;<lpage>1511</lpage>. <pub-id pub-id-type="doi">10.1002/cmdc.201300527</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bertoli</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Marabotti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2007b</year>). <article-title>Energetics of the protein-DNA-water interaction</article-title>. <source>BMC Struct. Biol.</source> <volume>7</volume>, <fpage>4</fpage>. <pub-id pub-id-type="doi">10.1186/1472-6807-7-4</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spyrakis</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Fornabaio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cozzini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mozzarelli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Computational titration analysis of a multiprotic HIV-1 protease ligand complex</article-title>. <source>J. Am. Chem. Soc.</source> <volume>126</volume>, <fpage>11764</fpage>&#x2013;<lpage>11765</lpage>. <pub-id pub-id-type="doi">10.1021/ja0465754</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bauer</surname>
<given-names>G. B.</given-names>
</name>
<name>
<surname>Kellogg</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Abraham</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Povirk</surname>
<given-names>L. F.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Effect of distamycin on chlorambucil-induced mutagenesis in pZ189: Evidence of a role for minor groove alkylation at adenine N-3</article-title>. <source>Mutagenesis</source> <volume>9</volume>, <fpage>133</fpage>&#x2013;<lpage>139</lpage>. <pub-id pub-id-type="doi">10.1093/mutage/9.2.133</pub-id>
</citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yakovchuk</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Protozanova</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Frank-Kamenetskii</surname>
<given-names>M. D.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Base-stacking and base-pairing contributions into thermal stability of the DNA double helix</article-title>. <source>Nucleic Acids Res.</source> <volume>34</volume>, <fpage>564</fpage>&#x2013;<lpage>574</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkj454</pub-id>
</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yilmazer</surname>
<given-names>N. D.</given-names>
</name>
<name>
<surname>Korth</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Recent progress in treating protein&#x2013;ligand interactions with quantum-mechanical methods</article-title>. <source>Int. J. Mol. Sci.</source> <volume>17</volume>, <fpage>742</fpage>. <pub-id pub-id-type="doi">10.3390/ijms17050742</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Web-based toolkits for topology prediction of transmembrane helical proteins, fold recognition, structure and binding scoring, folding-kinetics analysis and comparative analysis of domain combinations</article-title>. <source>Nucleic Acids Res.</source> <volume>33</volume>, <fpage>W193</fpage>&#x2013;<lpage>W197</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gki360</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>