<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioinform.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Bioinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioinform.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2673-7647</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1630078</article-id>
<article-id pub-id-type="doi">10.3389/fbinf.2025.1630078</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>COC<inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA - a fast and scalable algorithm for interatomic contact detection in proteins using C<inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distance matrices</article-title>
<alt-title alt-title-type="left-running-head">Lemos et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbinf.2025.1630078">10.3389/fbinf.2025.1630078</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lemos</surname>
<given-names>Rafael Pereira</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2416116"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mariano</surname>
<given-names>Diego</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1250353"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Silveira</surname>
<given-names>Sabrina De Azevedo</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1251302"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>de Melo-Minardi</surname>
<given-names>Raquel C.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1147412"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution>Laboratory of Bioinformatics and Systems, Department of Computer Science, Federal University of Minas Gerais</institution>, <city>Belo Horizonte</city>, <country country="BR">Brazil</country>
</aff>
<aff id="aff2">
<label>2</label>
<institution>Laboratory of Bioinformatics, Visualization and Systems, Department of Informatics, Federal University of Vi&#xe7;osa</institution>, <city>Vi&#xe7;osa</city>, <country country="BR">Brazil</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Rafael Pereira Lemos, <email xlink:href="rafaellemos@ufmg.br">rafaellemos@ufmg.br</email>
</corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-09-01">
<day>01</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>5</volume>
<elocation-id>1630078</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Lemos, Mariano, Silveira and de Melo-Minardi.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Lemos, Mariano, Silveira and de Melo-Minardi</copyright-holder>
<license>
<ali:license_ref start_date="2025-09-01">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>Protein interatomic contacts, defined by spatial proximity and physicochemical complementarity at atomic resolution, are fundamental to characterizing molecular interactions and bonding. Methods for calculating contacts are generally categorized as cutoff-dependent, which rely on Euclidean distances, or cutoff-independent, which utilize Delaunay and Voronoi tessellations. While cutoff-dependent methods are recognized for their simplicity, completeness, and reliability, traditional implementations remain computationally expensive, posing significant scalability challenges in the current Big Data era of bioinformatics. Here, we introduce COC<inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA (COntact search pruning by C<inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> Distance Analysis), a Python-based command-line tool for improving search pruning in large-scale interatomic protein contact analysis using alpha-carbon (C<inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) distance matrices. COC<inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA detects intra- and inter-chain contacts, and classifies them into seven different types: hydrogen and disulfide bonds; hydrophobic effects; attractive, repulsive, and salt-bridge interactions; and aromatic stackings. To evaluate our tool, we compared it with three traditional approaches in the literature: all-against-all atom distance calculation (&#x201c;brute-force&#x201d;), static C<inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distance cutoff (SC), and Biopython&#x2019;s NeighborSearch class (NS). COC<inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA demonstrated superior performance compared to the other methods, achieving on average 6x faster computation times than advanced data structures like <italic>k</italic>-d trees from NS, in addition to being simpler to implement and fully customizable. The presented tool facilitates exploratory and large-scale analyses of interatomic contacts in proteins in a simple and efficient manner, also enabling the integration of results with other tools and pipelines. The COC<inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA tool is freely available at <ext-link ext-link-type="uri" xlink:href="https://github.com/LBS-UFMG/COCaDA">https://github.com/LBS-UFMG/COCaDA</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>COC&#x3b1;DA</kwd>
<kwd>protein interactions</kwd>
<kwd>contacts</kwd>
<kwd>structural bioinformatics</kwd>
<kwd>command-line tool</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. This study was financed in part by the Coordena&#xe7;&#xe3;o de Aperfei&#xe7;oamento de Pessoal de N&#xed;vel Superior - Brasil (CAPES) - Finance Code 001; Funda&#xe7;&#xe3;o de Amparo &#xe0; Pesquisa do Estado de Minas Gerais - Brasil (FAPEMIG) - Finance Codes APQ-01834-21, APQ-02690-22, APQ-01838-24; and Conselho Nacional de Desenvolvimento Cient&#xed;fico e Tecnol&#xf3;gico - Brasil (CNPq) - Finance Codes 310406/2023-4, 440307/2022-8.</funding-statement>
</funding-group>
<counts>
<fig-count count="6"/>
<table-count count="2"/>
<equation-count count="6"/>
<ref-count count="37"/>
<page-count count="13"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Protein Bioinformatics</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Proteins are essential biological macromolecules, composed of amino acid residues linked by covalent peptide bonds. Their final three-dimensional structure is shaped not only by these covalent connections but also by weaker interactions such as hydrogen bonds, electrostatic forces, and hydrophobic effects (<xref ref-type="bibr" rid="B31">Smetana and Misra, 2017</xref>). The correct folding and stability of proteins are critical for their biological functions, making structural analysis fundamental for understanding cellular mechanisms, identifying therapeutic targets, and guiding the development of new drugs.</p>
<p>Since the first experimental resolution of a protein structure in 1958 (<xref ref-type="bibr" rid="B19">Kendrew et al., 1958</xref>), the field of structural biology has seen tremendous advances. Initiatives such as the Protein Data Bank (PDB, (<xref ref-type="bibr" rid="B5">Berman et al., 2000</xref>)) and, more recently, the AlphaFold Protein Structure Database (AFDB, (<xref ref-type="bibr" rid="B34">Varadi et al., 2024</xref>)) have centralized experimentally resolved and computationally predicted protein structures, making them widely accessible. The rapid growth of these repositories reflects not only advances in experimental techniques, such as X-ray crystallography, NMR spectroscopy, and cryo-electron microscopy, but also the impact of computational modeling approaches, including deep learning-based tools such as AlphaFold2 (<xref ref-type="bibr" rid="B18">Jumper et al., 2021</xref>) and AlphaFold3 (<xref ref-type="bibr" rid="B1">Abramson et al., 2024</xref>). These advances are part of the &#x201c;Big Data era in Bioinformatics&#x201d;, characterized by challenges related to data storage, processing, and interpretation at scale (<xref ref-type="bibr" rid="B24">Mura et al., 2018</xref>; <xref ref-type="bibr" rid="B25">Pal et al., 2020</xref>; <xref ref-type="bibr" rid="B23">Mariano et al., 2023</xref>).</p>
<p>The PDB currently holds 238,922 entries, with approximately 92% corresponding to protein structures<xref ref-type="fn" rid="n1">
<sup>1</sup>
</xref>. The archive continues to grow at an annual rate of around 6.5%<xref ref-type="fn" rid="n2">
<sup>2</sup>
</xref>, driven by both experimental and computational contributions (<xref ref-type="bibr" rid="B20">Kovalevskiy et al., 2024</xref>). This exponential expansion highlights the urgent need for computational strategies that can efficiently organize, validate, and analyze structural data at large scale. In particular, it is crucial to develop tools capable of supporting fundamental research, as well as applications in biomedical and biotechnological fields.</p>
<p>One key aspect of protein structure analysis is the characterization of interatomic contacts. Contacts are defined as spatial relationships between atoms or residues either within a molecule or between molecules, and are crucial for understanding protein-protein interactions, structural stability, and ligand binding mechanisms (<xref ref-type="bibr" rid="B10">da Silveira et al., 2009</xref>; <xref ref-type="bibr" rid="B27">Pires et al., 2011</xref>). In this context, it is important to distinguish between &#x201c;contacts&#x201d;, defined purely by spatial proximity, and &#x201c;interactions&#x201d;, which imply energetic contributions such as hydrophobic or electrostatic forces (<xref ref-type="bibr" rid="B16">Godzik et al., 1992</xref>; <xref ref-type="bibr" rid="B10">da Silveira et al., 2009</xref>). While not every contact results in a functional interaction, the presence of contacts is often a prerequisite for biologically relevant interactions. Therefore, in the remainder of this paper, the terms contact and interaction may be used interchangeably where appropriate, with &#x201c;contact&#x201d; referring primarily to spatial proximity and &#x201c;interaction&#x201d; to biochemical context.</p>
<p>Computational methods for contact identification offer an efficient alternative to labor-intensive experimental approaches, facilitating large-scale analyses across protein families and databases (<xref ref-type="bibr" rid="B12">Ding and Kihara, 2018</xref>). Traditionally, contacts are identified using Euclidean distance thresholds or cutoff-independent methods such as Voronoi (<xref ref-type="bibr" rid="B35">Voronoi, 1908</xref>) or Delaunay tessellations (<xref ref-type="bibr" rid="B11">Delaunay, 1934</xref>). Although cutoff-independent approaches are more sophisticated in theory, distance-based methods are often preferred for their simplicity, efficiency, and interpretability (<xref ref-type="bibr" rid="B10">da Silveira et al., 2009</xref>; <xref ref-type="bibr" rid="B27">Pires et al., 2011</xref>). Recent refinements incorporate physicochemical characteristics such as polarity or charge alongside spatial proximity, improving the biological relevance of computational predictions and reducing the incidence of false positives.</p>
<p>Several tools and databases have been developed to identify and analyze protein contacts (<xref ref-type="bibr" rid="B37">Wallace et al., 1995</xref>; <xref ref-type="bibr" rid="B22">Mancini et al., 2004</xref>; <xref ref-type="bibr" rid="B28">Schreyer and Blundell, 2009</xref>; <xref ref-type="bibr" rid="B21">Laskowski and Swindells, 2011</xref>; <xref ref-type="bibr" rid="B6">Bickerton et al., 2011</xref>; <xref ref-type="bibr" rid="B27">Pires et al., 2011</xref>; <xref ref-type="bibr" rid="B29">Schreyer and Blundell, 2013</xref>; <xref ref-type="bibr" rid="B17">Jubb et al., 2017</xref>; <xref ref-type="bibr" rid="B14">Fassio et al., 2020</xref>; <xref ref-type="bibr" rid="B26">Pimentel et al., 2021</xref>). However, existing solutions often present one or more limitations: they may be static, based on predefined datasets; computationally expensive, hindering large-scale use; restricted by server bottlenecks; limited to specific contact types such as residue-residue or protein-ligand; based on cutoff-independent methods; unsupported for modern file formats such as mmCIF; or discontinued altogether.</p>
<p>While these algorithms are well-established in the literature and typically can perform well for single structures, their computational cost becomes a bottleneck in large-scale analyses. Although our current study is based on experimentally determined structures from the PDB, the underlying method is designed with scalability in mind. The landscape of available protein structures has been further expanded by ultra-large-scale prediction initiatives. Notably, the AFDB now provides access to millions of high-confidence predicted models, vastly increasing the volume of structural data available for analysis. This shift underscores the growing importance of methods that combine accuracy with computational efficiency, as the feasibility of analyzing such extensive datasets hinges on scalable algorithms. In addition, time-resolved techniques such as molecular dynamics (MD) simulations introduce another dimension of complexity. These simulations generate thousands of frames per trajectory, each representing a unique protein conformation. Performing contact calculations across such datasets requires algorithms that can process structural information repeatedly and efficiently.</p>
<p>In response to these challenges, we propose COC<inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA (COntact search pruning by C<inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> Distance Analysis), a novel, Python-based approach for efficient large-scale identification of inter- and intrachain atomic contacts in proteins. COC<inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA applies optimized contact cutoffs derived from a systematic analysis of all protein structures in the PDB, leveraging maximum C<inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distances to enhance accuracy and consistency. The tool features a customized parser capable of handling both PDB and mmCIF formats, offering options for large file management, residue and contact filtering, and geometric property calculations such as centroids and normal vectors for aromatic residues. To support scalability and flexibility, COC<inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA allows parallel batch processing across multiple CPU cores, and user-defined custom contact distances.</p>
<p>To validate and benchmark COC<inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA, we performed two case studies: a small-scale benchmark involving gold-standard enzyme superfamilies to compare against existing slower methods, and a large-scale application covering all PDB entries with fewer than 10,000 residues. These evaluations demonstrate COC<inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA&#x2019;s capacity for accurate, high-throughput structural analysis, opening new avenues for research in protein evolution, pathogen mutation tracking, virtual compound screening, and beyond. COC<inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA can also be easily adapted to any existing analysis workflow, or be run independently for exploratory purposes.</p>
</sec>
<sec sec-type="methods" id="s2">
<label>2</label>
<title>Methodology</title>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> outlines the methodology for developing and benchmarking COC<inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA. The process begins with defining contacts and applying a static cutoff distance to the full PDB dataset. COC<inline-formula id="inf19">
<mml:math id="m19">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA then uses the maximum possible C<inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distance matrix to improve contact detection. The tool was benchmarked against similar methods using two datasets, focusing on processing time and computational complexity.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Overview of the methodology used to create and benchmark COC<inline-formula id="inf21">
<mml:math id="m21">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA. Initial contact definitions were based on previously published studies. Along with a general implementation of contact detection using fixed cutoff distances, these conditions were applied to the full set of protein structures available in the PDB. This first step led to the creation of a distance matrix, resulting in the improved implementation called COC<inline-formula id="inf22">
<mml:math id="m22">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA. COC<inline-formula id="inf23">
<mml:math id="m23">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA was then compared to different methods available in the literature using two distinct datasets. Finally, the results were analyzed in terms of processing time and complexity, demonstrating that our tool outperforms its competitors.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fbinf-05-1630078-g001.tif">
<alt-text content-type="machine-generated">Flowchart illustrating the methodology used to create and benchmark COCaDA. It starts with contact definition from previous studies, applied to the PDB archive dataset using a static cutoff value, leading to the distance matrix creation. The distance matrix creates COCaDA (Contact Optimization by C-alpha Distance Analysis), which is benchmarked against different methods from the literature. Time and complexity analysis are evaluated, and COCaDA outperforms the others.</alt-text>
</graphic>
</fig>
<sec id="s2-1">
<label>2.1</label>
<title>Contact definition</title>
<p>To store the contact types and their conditions, we used a dictionary containing all heavy atoms from the 20 standard amino acids, as defined in (<xref ref-type="bibr" rid="B32">Sobolev et al., 1999</xref>; <xref ref-type="bibr" rid="B30">Silva et al., 2019</xref>; <xref ref-type="bibr" rid="B14">Fassio et al., 2020</xref>; <xref ref-type="bibr" rid="B2">Barroso et al., 2020</xref>; <xref ref-type="bibr" rid="B26">Pimentel et al., 2021</xref>; <xref ref-type="bibr" rid="B13">Dos Santos et al., 2022</xref>). All 20 standard amino acids had their heavy atoms classified by the following characteristics, in binary form (<xref ref-type="table" rid="T1">Table 1</xref>): tendency to contribute to hydrophobic effects, belonging to aromatic groups, having positive charge, having negative charge, capability of donating or accepting electrons. The full atom classification table is available in the <xref ref-type="sec" rid="s11">Supplementary Table S1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Example of the binary classification of heavy atoms, according to their characteristics. For each amino acid residue, all their heavy atoms were classified in a binary manner, according to the following characteristics (hydrophobic, aromatic, positive, negative, donor, acceptor). Atom names follow the PDB nomenclature.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Residue</th>
<th align="left">Atom</th>
<th align="center">Hydrophobic</th>
<th align="center">Aromatic</th>
<th align="center">Positive</th>
<th align="center">Negative</th>
<th align="center">Donor</th>
<th align="center">Acceptor</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Alanine</td>
<td align="left">N</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">1</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">Arginine</td>
<td align="left">NH2</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">1</td>
<td align="center">0</td>
<td align="center">1</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">Glutamate</td>
<td align="left">OD1</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">1</td>
<td align="center">0</td>
<td align="center">1</td>
</tr>
<tr>
<td align="left">Glycine</td>
<td align="left">CA</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">0</td>
</tr>
<tr>
<td align="left">Tryptophan</td>
<td align="left">CZ2</td>
<td align="center">1</td>
<td align="center">1</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">0</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The possible contact types are: hydrogen and disulfide bonds; hydrophobic effects; attractive, repulsive, and salt bridge interactions; and aromatic stackings. This dictionary also contains the conditions needed for the contact (e.g., to form an attractive interaction, the atoms must be differently charged), and the range of Euclidean distances, in angstroms, for the contact to occur (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Summary of Types, Range and Conditions for contacts to occur. <inline-formula id="inf24">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; Euclidean distance between the atom pair.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Contact type</th>
<th align="center">Range (&#xc5;)</th>
<th align="center">Condition (other than range)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Hydrogen Bond</td>
<td align="left">
<inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>a</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>3.9</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Acceptor &#x2b; Donor atoms</td>
</tr>
<tr>
<td align="center">Disulfide Bond</td>
<td align="left">
<inline-formula id="inf26">
<mml:math id="m26">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>a</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>2.8</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Cys:SG &#x2b; Cys:SG atoms</td>
</tr>
<tr>
<td align="center">Hydrophobic</td>
<td align="left">
<inline-formula id="inf27">
<mml:math id="m27">
<mml:mrow>
<mml:mn>2.0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>a</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>4.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Hydrophobic &#x2b; Hydrophobic atoms</td>
</tr>
<tr>
<td align="center">Repulsive</td>
<td align="left">
<inline-formula id="inf28">
<mml:math id="m28">
<mml:mrow>
<mml:mn>2.0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>a</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>6.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Equally charged atoms</td>
</tr>
<tr>
<td align="center">Attractive</td>
<td align="left">
<inline-formula id="inf29">
<mml:math id="m29">
<mml:mrow>
<mml:mn>3.9</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>a</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>6.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Differently charged atoms</td>
</tr>
<tr>
<td align="center">Salt Bridge</td>
<td align="left">
<inline-formula id="inf30">
<mml:math id="m30">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>a</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>3.9</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Equally charged atoms &#x2b; hydrogen bonding</td>
</tr>
<tr>
<td align="center">Aromatic Stacking</td>
<td align="left">
<inline-formula id="inf31">
<mml:math id="m31">
<mml:mrow>
<mml:mn>2.0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>a</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>5.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Centroids of two aromatic rings in parallel or perpendicular orientation</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-2">
<label>2.2</label>
<title>Protein Data Bank archive</title>
<p>The full PDB protein archive, in &#x2018;.cif&#x2019; format, was obtained using in-house scripts to query and download entries directly from the PDB API. First, a script was used to query the API for entries containing &#x201c;Protein&#x201d; as an exact match from the parameter &#x201c;entity_poly.rcsb_entity_polymer_type&#x201d;.</p>
<p>To avoid rate limits and overwhelming the server, queries had a 1 s delay from one another, and only 25,000 IDs were obtained at a time. Then, a second script was used, together with the Biopython Bio. PDB module (<xref ref-type="bibr" rid="B9">Cock et al., 2009</xref>), to download all files that matched the IDs gathered in the first step. All files were downloaded between July 4th and 10 July 2024.</p>
</sec>
<sec id="s2-3">
<label>2.3</label>
<title>Neighbor search implementation using biopython</title>
<p>To serve as a comparison to our method, the Biopython package (<xref ref-type="bibr" rid="B9">Cock et al., 2009</xref>), largely used in bioinformatics, was used. The Bio. PDB module contains tools to parse a. pdb or. cif file, as well as the NeighborSearch (NS) class, which is useful in interatomic contact determination.</p>
<p>We used an in-house Python script to perform an all-atom neighbor search of 6&#xc5; radius, the maximum distance for contacts defined in our dictionary. Then, the neighbors were filtered based on their distance and physicochemical properties relative to the parent atom. Redundant comparisons were excluded; for example, if atom &#x201c;<inline-formula id="inf32">
<mml:math id="m32">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>&#x201d; was identified as a neighbor of atom &#x201c;<inline-formula id="inf33">
<mml:math id="m33">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>,&#x201d; the comparison was performed only once, preventing the redundant evaluation of &#x201c;<inline-formula id="inf34">
<mml:math id="m34">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>&#x201d; as a neighbor of &#x201c;<inline-formula id="inf35">
<mml:math id="m35">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.&#x201d; The code for the NS implementation is available at the Supplementary GitHub repository.</p>
<p>The contacts obtained contained the following information: chain, residue number, and parent atom name; chain, residue number, and neighbor atom name (i.e., the atomic pair making the contact); type of contact; and distance between the two atoms.</p>
</sec>
<sec id="s2-4">
<label>2.4</label>
<title>General implementation</title>
<p>To analyze the PDB protein archive and obtain the maximum distances matrix used in the rest of this work, we first devised a Static Cutoff (SC) implementation, where the C<inline-formula id="inf36">
<mml:math id="m36">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> cutoff distance was fixed. Akin to Biopython, proteins are treated as Python objects, containing chains, residues, and atoms. The package includes a customized. pdb/.cif parser, optimized to rapidly extract only the information relevant for contact determination. This makes it more efficient and lightweight than general-purpose parsers, which are typically designed to support a broader range of structural analysis tasks. By default, the parser considers the following criteria: a) Only the first model of each protein is considered (in the case of proteins experimentally resolved by NMR); b) Only atoms with occupancy <inline-formula id="inf37">
<mml:math id="m37">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.50 are considered; c) Water molecules, hydrogen atoms, non-standard residues, nucleic acids (DNA and RNA), and metallic coordination are not considered.</p>
<p>After parsing, the protein object is passed to a contact calculation script, where the C<inline-formula id="inf38">
<mml:math id="m38">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distances for each pair of residues are obtained, and filtered based on the fixed cutoff. Centroids of aromatic rings were calculated using all atoms belonging to the ring, and the calculation of normal vectors and angles was performed using the Python NumPy library.</p>
<p>The atoms from the residues that are in range to interact are then compared to the dictionary previously described, based on their distance to each other, and their physicochemical properties. Finally, the contacts are returned in a custom object containing all their information, similar to the NS method.</p>
</sec>
<sec id="s2-5">
<label>2.5</label>
<title>Distance matrix</title>
<p>Throughout the processing of the complete PDB archive using the SC method, the maximum distances (across all proteins in the PDB) between the C<inline-formula id="inf39">
<mml:math id="m39">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> atoms of each amino acid pair were stored in a distance matrix. Upon completion of the processing, this distance matrix was then used to update the static cutoff point employed in the SC method, generating specific values for each amino acid pair.</p>
<p>The distance matrix <inline-formula id="inf40">
<mml:math id="m40">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a square matrix of size <inline-formula id="inf41">
<mml:math id="m41">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf42">
<mml:math id="m42">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of standard amino acids. Each entry <inline-formula id="inf43">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> corresponds to the maximum distance between the C<inline-formula id="inf44">
<mml:math id="m44">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> atoms of the amino acids at positions <inline-formula id="inf45">
<mml:math id="m45">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf46">
<mml:math id="m46">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (e.g., <inline-formula id="inf47">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>11</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents an Alanine pair, and <inline-formula id="inf48">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents a Valine pair):<disp-formula id="e1">
<mml:math id="m49">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>11</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>12</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>21</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>22</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x22f1;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where each <inline-formula id="inf49">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the maximum Euclidean distance between the C<inline-formula id="inf50">
<mml:math id="m51">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> atoms of the amino acids at positions <inline-formula id="inf51">
<mml:math id="m52">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf52">
<mml:math id="m53">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>A total of 210 distance values were obtained, representing each possible residue pair and excluding redundancies (e.g., Ala-Val is the same as Val-Ala) (<xref ref-type="disp-formula" rid="e2">Equation 2</xref>).<disp-formula id="e2">
<mml:math id="m54">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf53">
<mml:math id="m55">
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of non-redundant distance pairs, and <inline-formula id="inf54">
<mml:math id="m56">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of standard amino acids. In this case, as <inline-formula id="inf55">
<mml:math id="m57">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 20, then <inline-formula id="inf56">
<mml:math id="m58">
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 210.</p>
</sec>
<sec id="s2-6">
<label>2.6</label>
<title>COC<inline-formula id="inf57">
<mml:math id="m59">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA implementation</title>
<p>The COC<inline-formula id="inf58">
<mml:math id="m60">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA tool is implemented similarly to the previous implementations (NS and SC). A schematic representation of the implementation is presented in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Scripts used for the COC<inline-formula id="inf59">
<mml:math id="m61">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA implementation. The steps for processing the command-line parameters (1), processing the input files and creating the objects (2), and calculating contacts using the values obtained from the distance matrix (3) are shown.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fbinf-05-1630078-g002.tif">
<alt-text content-type="machine-generated">Flowchart showing the scripts used for COCaDA implementation. It initializes with &#x22;main.py&#x22;, which references three steps: 1) &#x22;argparser.py&#x22; for file paths, CSV output, parallel processing, and custom distances. 2) &#x22;parser.py&#x22; processes PDB/CIF files into "classes.py" covering protein, chain, residue, and atom. 3) &#x22;conditions.py&#x22; and &#x22;distances.py&#x22; lead to &#x22;contacts.py&#x22;.</alt-text>
</graphic>
</fig>
<p>First, the script &#x201c;main.py&#x201d; is executed via the command line, along with the required parameters, which are processed by the script &#x201c;argparser.py&#x201d; (1). The parameters include the mandatory file paths (wildcards are accepted), an optional binary flag for generating an output file in. csv format (default &#x3d; no), an optional parameter to parallelize file processing in batches across any combination of available CPU cores, and an optional parameter to use custom contact distances defined by the user instead of those defined in <xref ref-type="table" rid="T2">Table 2</xref> (using the &#x201c;contact_distances.json&#x201d; configuration file). All parameters are fully explained using the &#x2018;-h&#x2019; or &#x2018;&#x2013;help&#x2019; flags.</p>
<p>When users specify custom contact cutoffs greater than the default (6&#xc5;), the static C<inline-formula id="inf60">
<mml:math id="m62">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>&#x2013;C<inline-formula id="inf61">
<mml:math id="m63">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distance matrix is extended accordingly. This is done by computing an epsilon <inline-formula id="inf62">
<mml:math id="m64">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> value, which is the difference between the user-defined cutoff and the default maximum cutoff. If <inline-formula id="inf63">
<mml:math id="m65">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, it is added to all distance values in the original matrix. This adjustment ensures that the pruning based on C<inline-formula id="inf64">
<mml:math id="m66">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distances remains valid even under relaxed contact definitions, preserving consistency with the originally computed matrix while accommodating user-defined thresholds.</p>
<p>The files are then processed by the script &#x201c;parser.py&#x201d; (2), which utilizes the previously defined classes to create objects representing proteins, their chains, residues, and atoms. Finally, the script &#x201c;contacts.py&#x201d; receives the objects generated by the parser (3). The scripts &#x201c;conditions.py&#x201d;, which stores the conditions required for contact detection, and &#x201c;distances.py&#x201d;, which stores the values obtained from the distance matrix, are used to compute the contacts.</p>
<p>If the user does not specify the. csv output file parameter, only a summary is displayed in the terminal, containing the protein name, residue count, number of contacts, and processing time in seconds. If the parameter is used, in addition to the summary, a. csv file is generated with detailed information about each detected protein contact. The columns in the output file are organized as follows: Chain 1, Residue Number 1, Residue Name 1, Atom Name 1, Chain 2, Residue Number 2, Residue Name 2, Atom Name 2, Distance, Contact Type. An example output file for PDB ID 101M, as well as a PyMOL (Schr&#xf6;dinger, LLC) visualization script to help users quickly explore the results, are available at the Supplementary GitHub repository.</p>
</sec>
<sec id="s2-7">
<label>2.7</label>
<title>Datasets</title>
<p>Two datasets were selected to benchmark our results and compare them to other competitors (<inline-formula id="inf65">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">dataset1</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 896 and <inline-formula id="inf66">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">dataset2</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 215,716). The first (D1) is a modified gold-standard set of enzyme superfamilies (<xref ref-type="bibr" rid="B8">Brown et al., 2006</xref>), with 365 unique entries ranging from 194 to 6,208 residues. For a more balanced comparison, we split all chains in different files, and treated them separately. The new modified dataset contains 896 entries, ranging from one to 994 residues (available on the Supplementary GitHub repository). The second dataset (D2) includes all PDB proteins with less than 10,000 residues, covering approximately 99.2% of all protein entries. This dataset contains 215,716 unique entries, ranging from three to 10,000 residues.</p>
</sec>
<sec id="s2-8">
<label>2.8</label>
<title>Benchmarks</title>
<p>To ensure fairness and eliminate biases, all benchmarks were conducted simultaneously on a server with the following specifications: NVIDIA A100 GPU, 768 GB RAM, and a 128-thread AMD Ryzen Threadripper 5995WX processor. To prevent memory overload and parallelization issues, each process was executed on an individual core. Due to its size, dataset D2 was divided into nine batches of approximately 25,000 files each, with each batch being processed independently on separate cores.</p>
<p>Although multithreading and batch processing is available for all implementations (NS, SC, and COC<inline-formula id="inf67">
<mml:math id="m69">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA) using the Python module &#x201c;concurrent.futures&#x201d;, each core handled only a distinct batch to maintain consistency in the results. The total processing time for each entry was defined as the sum of the file reading, parsing, contact detection, and output generation times.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<label>3</label>
<title>Results and discussion</title>
<sec id="s3-1">
<label>3.1</label>
<title>Maximum distance matrix</title>
<p>In total, 217,454 PDB entries were downloaded in. cif format, totaling approximately 450 GB (The full ID list is available on the Supplementary GitHub repository). Proteins ranged from three (PDB IDs: 1Q7O, 8DDG, 8DDH) to 503,221 (PDB ID: 8GLV) modeled amino acid residues. To obtain the values for the distance matrix, we processed all the downloaded files using a fixed C<inline-formula id="inf68">
<mml:math id="m70">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distance cutoff of 21&#xc5; for all pairs of residues (SC). This value is comfortably above the maximum distance between the C<inline-formula id="inf69">
<mml:math id="m71">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of a pair of arginines, the biggest residues by length, that are able to have contacts between their side-chain atoms (considering a maximum contact distance of 6&#xc5;, as per <xref ref-type="table" rid="T2">Table 2</xref>). To confirm this, we compared an all-atom approach (i.e., comparing every atom of the protein against each other, without cutoffs) to the SC approach using D1, and no contacts were missed (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>).</p>
<p>Using the SC implementation and the 217,454 entries downloaded from the PDB, over 211 million amino acid residues and 819 million contacts were processed and identified. Along this process, we stored the maximum C<inline-formula id="inf70">
<mml:math id="m72">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distances for every pair of the 20 standard amino acids, and after merging redundancies, we obtained 210 values in a symmetric distance matrix (<xref ref-type="fig" rid="F3">Figure 3</xref>; <xref ref-type="disp-formula" rid="e1">Equation 1</xref>). The full distance table is available in the <xref ref-type="sec" rid="s11">Supplementary Table S3</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Distance Matrix between the C<inline-formula id="inf71">
<mml:math id="m73">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of all pairs of residues. The intensity of the color indicates the scale of the value (from 7 to 20 &#xc5;). The highlighted diagonal represents pairs of the same residue (e.g., Ala-Ala). The full list of values is available in the <xref ref-type="sec" rid="s11">Supplementary Table S3</xref>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fbinf-05-1630078-g003.tif">
<alt-text content-type="machine-generated">A heatmap displaying distances the maximum alpha-carbon distances between amino acid pairs, labeled along the axes with abbreviations like ALA, ARG, ASN, etc. Shades range from light to dark blue, representing distances from 8 to 20 angstroms. The color scale is shown on the right.</alt-text>
</graphic>
</fig>
<p>As the distance matrix is color-coded based on the value of the maximum C<inline-formula id="inf72">
<mml:math id="m74">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distance, we can quickly spot the minimum and maximum values obtained. For the lowest value encountered, we found a pair consisting of an alanine and a glycine residue, both present in the HD chain of PDB ID 6QCM, with a distance of 7.65&#xc5; between their C<inline-formula id="inf73">
<mml:math id="m75">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>&#x2019;s (<xref ref-type="fig" rid="F4">Figure 4A</xref>). This is expected, as alanines and glycines are two of the smallest amino acid residues, differing only by a single <inline-formula id="inf74">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CH</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> group in the side chain of the alanine, while glycine has a hydrogen atom in its side chain. However, even with this difference, the presence of the <inline-formula id="inf75">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CH</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> group on the side chain of the alanine does not impact the distance between their C<inline-formula id="inf76">
<mml:math id="m78">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> atoms, only contributing to the chirality of the alanine residue. For these two residues, only hydrogen bonds are possible, as the main chain atoms are only capable of donating (main chain nitrogen) or accepting (main chain oxygen) hydrogen atoms.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Minimum and maximum entries in the Distance Matrix. <bold>(A)</bold> Minimum value, in a hydrogen bond between a glycine and an alanine residue. <bold>(B)</bold> Maximum value, in a repulsive interaction between two arginine residues. The higher number (larger dotted line) represents the distance between C<inline-formula id="inf77">
<mml:math id="m79">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and the lower one (smaller dotted line) the contact distance. The PDB IDs are shown in center, and the contact details are shown in the format Chain:Residue-Atom.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fbinf-05-1630078-g004.tif">
<alt-text content-type="machine-generated">Diagram showing residue pairs with labeled atoms and distances. Part A features PDB ID 6QCM, containing the lowest values, with distances of 3.89 angstroms and 7.65 angstroms between the pair HD:G188 and HD:A221. Part B shows PDB ID 3X0Y, containing the highest values, with distances of 5.90 angstroms and of 20.46 angstroms between the pair H:R211 and G:R211. Elements are color-coded: nitrogen in blue, carbon in gray, and oxygen in red. Dashed lines indicate measured distances.</alt-text>
</graphic>
</fig>
<p>On the other hand, a pair of two arginine residues, from chains G and H of PDB ID 3X0Y, represents the highest C<inline-formula id="inf78">
<mml:math id="m80">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distance encountered, of 20.46&#xc5; (<xref ref-type="fig" rid="F4">Figure 4B</xref>). This result is also expected and consistent with the fixed distance of 21&#xc5; used in the SC approach, once again demonstrating that the fixed cutoff was appropriate to yield no missed contacts. The contact itself is a repulsive interaction between two <inline-formula id="inf79">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>NH</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> atoms from the residues side-chains, at a distance of 5.9&#xc5;.</p>
<p>If the user wishes to apply custom contact distances instead of those defined in <xref ref-type="table" rid="T2">Table 2</xref>, the optional &#x2018;-d&#x2019; flag can be used, specifying the desired values in the &#x201c;contact_distances.json&#x201d; configuration file. This flag extends the C<inline-formula id="inf80">
<mml:math id="m82">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distance values in the distance matrix to incorporate the user-defined distances, ensuring that no contacts are omitted.</p>
<p>Use-case scenarios for modifying the default cutoff distances include adopting alternative minimum or maximum values reported in the literature, such as 6&#xc5; for aromatic stacking (<xref ref-type="bibr" rid="B15">Fassio et al., 2022</xref>), or 5&#xc5; for hydrophobic effects (<xref ref-type="bibr" rid="B6">Bickerton et al., 2011</xref>), instead of the default 5&#xc5; and 4.5&#xc5;, respectively. Another common scenario involves exploratory analyses using step-wise cutoff variations (<xref ref-type="bibr" rid="B10">da Silveira et al., 2009</xref>; <xref ref-type="bibr" rid="B33">Vangone and Bonvin, 2015</xref>).</p>
</sec>
<sec id="s3-2">
<label>3.2</label>
<title>COC<inline-formula id="inf81">
<mml:math id="m83">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA</title>
<p>With the maximum possible C<inline-formula id="inf82">
<mml:math id="m84">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distances properly established for all amino acid residue pairs, we updated with the new values the &#x201c;distances&#x201d; dictionary from the SC implementation (<xref ref-type="sec" rid="s2-4">Section 2.4</xref>), which before was fixed at 21&#xc5; for all residue pairs. To the joint implementation of the SC method with the distances updated from the distance matrix, we gave the name COC<inline-formula id="inf83">
<mml:math id="m85">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA.</p>
<p>By using tightly-defined, pair-specific cutoff distances, we can further enhance the search space pruning compared to using a single fixed distance threshold. Analysis of the distribution of maximum C<inline-formula id="inf84">
<mml:math id="m86">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distances reveals that most amino acid pairs exhibit distances well below 21&#xc5;, suggesting that introducing pairwise-specific cutoffs is justified and efficient (<xref ref-type="sec" rid="s11">Supplementary Figures S1 and S2</xref>). Moreover, given that dictionary lookups in Python operate with linear average-case complexity (O(1)), storing and querying 210 unique cutoff values introduces negligible computational overhead relative to using a single fixed value.</p>
<p>To improve efficiency, COC<inline-formula id="inf85">
<mml:math id="m87">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA employs a stepwise filtering strategy that first evaluates residue-level proximity based on C<inline-formula id="inf86">
<mml:math id="m88">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>&#x2013;C<inline-formula id="inf87">
<mml:math id="m89">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distances. An initial coarse filter excludes residue pairs exceeding the global maximum cutoff (20.46&#xc5;, for Arg&#x2013;Arg pairs), allowing early removal of clearly non-interacting pairs. Remaining pairs are then subjected to a more stringent, pair-specific cutoff comparison using the distance matrix. Only those pairs that satisfy both criteria proceed to atomic-level evaluation to determine whether a contact is present. This tiered approach reduces the number of atomic comparisons required, helping to balance computational cost with contact detection accuracy. A schematic illustration of this process is provided in the <xref ref-type="sec" rid="s11">Supplementary Figure S3</xref>.</p>
<p>The residue-pair-specific distance matrix used in COC<inline-formula id="inf88">
<mml:math id="m90">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA captures the full range of C<inline-formula id="inf89">
<mml:math id="m91">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>&#x2013;C<inline-formula id="inf90">
<mml:math id="m92">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distances observed across known protein structures that are compatible with side-chain atomic contacts. Because these thresholds are grounded in the physical constraints that govern residue-residue contacts, the method is broadly applicable regardless of the overall fold, resolution, or origin of the structural model. Any contact that can realistically occur must still conform to these spatial constraints, ensuring generalization across diverse structural contexts.</p>
<p>While the method is robust to structural variability at the backbone level, the accuracy of contact detection may be influenced by uncertainty in side-chain atom positions, particularly in lower-resolution or flexible regions. In such cases, performance near cutoff boundaries may be affected, which is an inherent limitation of any deterministic cutoff-dependent method.</p>
</sec>
<sec id="s3-3">
<label>3.3</label>
<title>Benchmarks</title>
<p>To benchmark COC<inline-formula id="inf91">
<mml:math id="m93">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA against other approaches used in the literature, we selected the following: all atoms against all atoms (AllAtoms, used in (<xref ref-type="bibr" rid="B26">Pimentel et al., 2021</xref>)), Arpeggio (<xref ref-type="bibr" rid="B17">Jubb et al., 2017</xref>), Arpeggio CLI<xref ref-type="fn" rid="n3">
<sup>3</sup>
</xref> (<xref ref-type="bibr" rid="B17">Jubb et al., 2017</xref>), Biopython Neighbor Search (NS, used in (<xref ref-type="bibr" rid="B6">Bickerton et al., 2011</xref>)), and Static Cutoff (SC). Other methods, like nAPOLI (<xref ref-type="bibr" rid="B14">Fassio et al., 2020</xref>), STING Contacts (<xref ref-type="bibr" rid="B22">Mancini et al., 2004</xref>), and PICCOLO (<xref ref-type="bibr" rid="B6">Bickerton et al., 2011</xref>), were not available at the time of search and were not updated recently, so they were not considered.</p>
<p>Both Arpeggio versions were too slow to process even small proteins, as our tests showed processing times of approximately 5 and 23 min for a single 1,000 residue protein (PDB ID 6RTH) for Arpeggio CLI and Arpeggio Web, respectively. For comparison, the same protein was processed in 0.62s using COC<inline-formula id="inf92">
<mml:math id="m94">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA. This can be due to several factors, but we believe that the explanation lies mainly in server load (for Arpeggio Web), and the several external libraries and computing time that are needed to run the more complex analysis (for both versions). In this way, since a large-scale analysis would not be feasible due to the large processing times, both versions (Web and CLI) were disregarded in the subsequent analyses.</p>
<p>For the AllAtoms approach, a new implementation was developed in Python, in order to incorporate all the previously defined definitions and constraints, and for the NS method, a custom implementation was created using Biopython (<xref ref-type="bibr" rid="B9">Cock et al., 2009</xref>), since the PICCOLO tool is currently unavailable. Thus, for D1, the following methods were compared: AllAtoms, NS, SC and COC<inline-formula id="inf93">
<mml:math id="m95">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA.</p>
<p>The first dataset contains 896 entries, ranging from one to 994 residues. The initial choice of a smaller dataset was made to include the AllAtoms approach, which is significantly slower than the other three (but still considerably faster than both versions of Arpeggio), which would make a large-scale analysis unfeasible.</p>
<p>In <xref ref-type="fig" rid="F5">Figure 5</xref>, it is possible to see that the AllAtoms approach (orange) rapidly explodes in a quadratic curve compared to the three others, which maintain rather linear calculation times up to 1,000 residues. Once again, no contacts of any type were missed in any of the approaches (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>), but the AllAtoms approach was removed from further analysis because of its performance. Comparing the faster approaches, SC (yellow) obtained calculation times 1.5x faster on average than NS (magenta), while COC<inline-formula id="inf94">
<mml:math id="m96">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA (cyan) showed calculation times 3.8x faster on average, obtaining the same contacts.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Protein Size vs. Computation Time plot of Benchmark 1. In the first benchmark, 896 files ranging from 1 to 994 residues were analyzed. COC<inline-formula id="inf95">
<mml:math id="m97">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA is shown in cyan, SC in yellow, NS in magenta, and AllAtoms in orange. Points represent individual entries, with lines showing the fitted curves for the data.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fbinf-05-1630078-g005.tif">
<alt-text content-type="machine-generated">Scatter plot showing computation time in seconds versus protein size in residues, for Dataset 1, for four methods: COCaDA (cyan), Static Cutoff (yellow), Neighbor Search (magenta), and All Atoms (orange), each fitted with a trend line. Time increases with protein size; All Atoms show the steepest increase. </alt-text>
</graphic>
</fig>
<p>As the results from the first, small dataset showed a significant difference in processing times between the 3 fastest approaches, we then moved to D2, which contains 215,716 unique entries, ranging from three to 10,000 residues, making approximately 99.2% of the PDB protein archive. The choice to remove entries above 10,000 residues was made due to the nature of those entries, which are mostly protein complexes, containing several copies of each unique chain. This makes them not suitable for contact analysis directly, requiring some kind of pre-processing, like splitting only the unique chains or working with each individual protein present in the complex separately. This can also be true for entries below 10,000 residues, but we believe that this slice correctly represents the diversity of experimentally resolved protein structures.</p>
<p>
<xref ref-type="fig" rid="F6">Figure 6</xref> shows the results for D2 when comparing the COC<inline-formula id="inf96">
<mml:math id="m98">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA (cyan), SC (yellow), and NS (magenta) approaches. It is possible to see that COC<inline-formula id="inf97">
<mml:math id="m99">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA performs better for all proteins, averaging approximately 6x faster times than NS and 2.5x faster times than SC. The SC approach performs better than NS in proteins below 5,000 residues, equal between 5,000 and 7,000 residues, and worse above 7,000 residues. However, since the vast majority of PDB entries fall within the smaller size range (approximately 97.2% of unique entries have 6,000 or fewer residues), where the performance gain of COC<inline-formula id="inf98">
<mml:math id="m100">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA and SC over NS is most pronounced, the high density of small proteins in the dataset significantly skews the overall average, leading to the reported 6x and 2.5x improvement.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Protein Size vs. Computation Time plot of Benchmark 2. In the second benchmark, 215,716 files were analyzed, ranging from 3 to 10,000 residues. COC<inline-formula id="inf99">
<mml:math id="m101">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA is shown in cyan, SC in yellow, and NS in magenta. A detail of the 0&#x2013;1,000 protein size range is shown in the upper left corner. Points represent individual entries, with lines showing the fitted curves for the data. Outliers were defined as <inline-formula id="inf100">
<mml:math id="m102">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 5 times the Standard Deviation for each approach.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fbinf-05-1630078-g006.tif">
<alt-text content-type="machine-generated">Scatter plot showing computation time in seconds versus protein size in residues, for Dataset 2, for three methods: COCaDA (cyan), Static Cutoff (yellow), and Neighbor Search (magenta), each fitted with a trend line. The graph includes an inset zooming in on the lower range of protein sizes. </alt-text>
</graphic>
</fig>
<p>Outliers were considered as entries that had a processing time <inline-formula id="inf101">
<mml:math id="m103">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 5 times the Standard Deviation for each approach, with less than 1% of entries removed. After outlier removal, it is possible to see that COC<inline-formula id="inf102">
<mml:math id="m104">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA has a consistent time vs. size distribution, while the other two approaches have more variation. This can be due to the tight and precise definition of cutoff distances for COC<inline-formula id="inf103">
<mml:math id="m105">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA, which speeds up a lot of the computation, while also limiting structural variations, like between globular and fibrillar proteins.</p>
<p>The protein size range of 0&#x2013;1,000 residues is noteworthy, as shown in detail in the upper left corner of <xref ref-type="fig" rid="F6">Figure 6</xref>. At this range, we can see the distribution pattern observed in D1, while also identifying several divergent entries in the NS approach. The divergent spike is composed exclusively of Nuclear Magnetic Resonance (NMR) resolved entries, which are usually deposited as several individual models of the same protein. Due to the nature of Biopython native parsing, all the models need to be parsed even if only the first one is of interest, unless the user creates specific functions for this purpose, thus deviating from the original implementation of the library. As these NMR entries are small, the parsing time of several NMR models outpaces the contact calculation time of the first one, leading to a spike in processing time. This does not occur in the COC<inline-formula id="inf104">
<mml:math id="m106">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA and SC approaches, as the customized parser handles only the selected model in the file (the default value is always the first model).</p>
</sec>
<sec id="s3-4">
<label>3.4</label>
<title>Empirical complexity analysis</title>
<p>Computing interatomic contacts is inherently a quadratic problem (O<inline-formula id="inf105">
<mml:math id="m107">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>) because it requires calculating the distance between every pair of atoms in a protein, such as in the AllAtoms approach. However, sophisticated data structures, such as <inline-formula id="inf106">
<mml:math id="m108">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-d trees, can be employed to avoid calculating distances between atoms/residues that are too far apart. This approach theoretically reduces the computational space by pruning irrelevant comparisons, leading to practical reductions in computation time and typically logarithmic (<inline-formula id="inf107">
<mml:math id="m109">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, in the average of cases) or linear (O<inline-formula id="inf108">
<mml:math id="m110">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, in the worst of cases) complexity (<xref ref-type="bibr" rid="B3">Bentley, 1975</xref>; <xref ref-type="bibr" rid="B4">Berg et al., 2008</xref>). However, these complexity values refer only to the operations of search, insertion, and deletion of elements in the trees. For the construction of a new tree&#x2014;for example, for a new protein&#x2014;the processing is substantially greater, corresponding to a log-linear complexity <inline-formula id="inf109">
<mml:math id="m111">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> in the best-case scenario, for balanced trees (<xref ref-type="bibr" rid="B36">Wald and Havran, 2006</xref>; <xref ref-type="bibr" rid="B7">Brown, 2015</xref>).</p>
<p>In the case of small entries, like most of the protein structures, the memory overhead associated with the allocation and creation of the tree usually does not outweigh the computational gains in search, insertion, and deletion operations. Thus, in addition to the high implementation complexity of <inline-formula id="inf110">
<mml:math id="m112">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-d trees, the results may turn out to be worse than those obtained by &#x2018;brute-force&#x2019; algorithms, provided that the latter are properly implemented. These gains are primarily concentrated in small proteins, which significantly contribute to the average observed speedup of 6x when comparing COC<inline-formula id="inf111">
<mml:math id="m113">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA to NS, even though the performance benefit becomes less pronounced on bigger proteins.</p>
<p>Our approach in COC<inline-formula id="inf112">
<mml:math id="m114">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA fits as an improvement of classical &#x2018;brute-force&#x2019; algorithms (such as AllAtoms) for the calculation of interatomic contacts in proteins. This is because, theoretically, all atoms of the protein are checked at least once, but the precise definitions of cutoff distances for each residue pair significantly reduce processing time, without the additional cost of using complex data structures, as is the case with <inline-formula id="inf113">
<mml:math id="m115">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-d trees.</p>
<p>In this study, we chose to evaluate the complexity of various algorithms empirically, by comparing standard methods commonly used in the structural bioinformatics community with the COC<inline-formula id="inf114">
<mml:math id="m116">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA method. These different methods were tested with inputs of increasing sizes (where <inline-formula id="inf115">
<mml:math id="m117">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of residues, which have on average eight atoms each), and we analyzed the resulting fitted curves with real datasets.</p>
<p>The curve fittings of the three approaches against the second dataset demonstrate that both COC<inline-formula id="inf116">
<mml:math id="m118">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA (<inline-formula id="inf117">
<mml:math id="m119">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> quadratic &#x3d; 0.99, <xref ref-type="disp-formula" rid="e3">Equation 3</xref>) and SC (<inline-formula id="inf118">
<mml:math id="m120">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> quadratic &#x3d; 0.97, <xref ref-type="disp-formula" rid="e4">Equation 4</xref>) exhibit quadratic growth trends, while NS shows a linear growth trend (<inline-formula id="inf119">
<mml:math id="m121">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> linear &#x3d; 0.97, <xref ref-type="disp-formula" rid="e5">Equation 5</xref>). This results demonstrate, experimentally, the nature of the contact identification functions, which are the most time-consuming operations. COC<inline-formula id="inf120">
<mml:math id="m122">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA and SC act basically as heavy improvements of &#x2018;brute force&#x2019; algorithms, having a time complexity of <inline-formula id="inf121">
<mml:math id="m123">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, while the NS contact identification function operates with a time complexity of <inline-formula id="inf122">
<mml:math id="m124">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, leading to its linear growth pattern.<disp-formula id="e3">
<mml:math id="m125">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.35</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>5.04</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6.36</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>;</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m126">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.20</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1.60</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>9.18</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>;</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m127">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2.37</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>7.94</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf123">
<mml:math id="m128">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf124">
<mml:math id="m129">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf125">
<mml:math id="m130">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are the best-fitted functions for COC<inline-formula id="inf126">
<mml:math id="m131">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA, SC, and NS, respectively, and <inline-formula id="inf127">
<mml:math id="m132">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of residues.</p>
<p>However, in the case of the NS approach, the memory overhead in tree construction is evident, especially for small inputs. It is possible to observe, in <xref ref-type="fig" rid="F5">Figure 5</xref>, that for proteins of up to 200 residues even the most computationally expensive approach (AllAtoms) achieves lower processing times than NS. This can also be seen in the range that includes small entries in D2, although with less detail due to the massive number of points (<xref ref-type="fig" rid="F6">Figure 6</xref>, detail).</p>
<p>The analysis of the coefficients from the obtained equations is another way to explain the poorer performance of NS, even though it grows linearly compared to the quadratic growth of the other approaches. Starting with the quadratic coefficients <inline-formula id="inf128">
<mml:math id="m133">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, it can be noted that both are practically insignificant (<inline-formula id="inf129">
<mml:math id="m134">
<mml:mrow>
<mml:mn>1.35</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for COC<inline-formula id="inf130">
<mml:math id="m135">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA and <inline-formula id="inf131">
<mml:math id="m136">
<mml:mrow>
<mml:mn>1.20</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for SC). This means that for small <inline-formula id="inf132">
<mml:math id="m137">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> values, as in D2, the quadratic growth is extremely slow, as can be seen from the slight curve in the data.</p>
<p>As for the linear coefficients <inline-formula id="inf133">
<mml:math id="m138">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, there is a significant difference between COC<inline-formula id="inf134">
<mml:math id="m139">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA <inline-formula id="inf135">
<mml:math id="m140">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>5.04</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and the other two approaches (<inline-formula id="inf136">
<mml:math id="m141">
<mml:mrow>
<mml:mn>1.60</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for SC and <inline-formula id="inf137">
<mml:math id="m142">
<mml:mrow>
<mml:mn>2.37</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for NS), with NS showing the highest value. This difference, along with the low quadratic coefficients, explains how NS&#x2019;s linear growth can be less efficient than the quadratic growth of the other approaches. Furthermore, as the SC and NS terms are close, between 6,000 and 7,000 residues the curves invert, representing the point where the quadratic growth of SC becomes more influential.</p>
<p>Finally, regarding the constant coefficients <inline-formula id="inf138">
<mml:math id="m143">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, again there is a marked difference between the three approaches, with SC having the smallest value <inline-formula id="inf139">
<mml:math id="m144">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>9.18</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, followed by COC<inline-formula id="inf140">
<mml:math id="m145">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA <inline-formula id="inf141">
<mml:math id="m146">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6.36</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, and NS <inline-formula id="inf142">
<mml:math id="m147">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>7.94</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, the only approach with a positive <inline-formula id="inf143">
<mml:math id="m148">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> value. This result highlights the memory overhead associated with <italic>k</italic>-d tree construction, which causes even small inputs to have a baseline processing time higher than the actual contact calculation time.</p>
<p>As previously shown, the NS approach (using <italic>k</italic>-d trees) has a linear time complexity in the worst case. Meanwhile, COC<inline-formula id="inf144">
<mml:math id="m149">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA exhibits quadratic growth, yet shows lower processing times for all analyzed entries. By equating the two equations (<xref ref-type="disp-formula" rid="e3">Equation 3</xref> for COC<inline-formula id="inf145">
<mml:math id="m150">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA; <xref ref-type="disp-formula" rid="e5">Equation 5</xref> for NS), we find that COC<inline-formula id="inf146">
<mml:math id="m151">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA would only yield worse results for proteins with approximately 14,000 residues (<xref ref-type="disp-formula" rid="e6">Equation 6</xref>).<disp-formula id="e6">
<mml:math id="m152">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mtext>when&#x2009;</mml:mtext>
<mml:mi>n</mml:mi>
<mml:mo>&#x2248;</mml:mo>
<mml:mn>14.000</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf147">
<mml:math id="m153">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> e <inline-formula id="inf148">
<mml:math id="m154">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are the best-fitted functions to represent COC<inline-formula id="inf149">
<mml:math id="m155">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA and NS data, respectively, and <inline-formula id="inf150">
<mml:math id="m156">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of residues.</p>
<p>However, only slightly more than 2,000 entries in the PDB have a size greater than 14,000 residues, representing less than 1% of all protein structures. Furthermore, all of these entries correspond to protein complexes or repetitions of the same protein, as previously discussed. Therefore, for all practical purposes of contact detection in proteins, the COC<inline-formula id="inf151">
<mml:math id="m157">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA approach demonstrates the best temporal performance compared to other methodologies in the literature.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<label>4</label>
<title>Conclusion</title>
<p>In a context where the influx of biological data is greater than ever, there is an increasing need for solutions that are efficient, robust, and scalable. In response to this demand, we developed COC<inline-formula id="inf152">
<mml:math id="m158">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA, a free, command-line tool designed to efficiently identify interatomic contacts in proteins at large scale. COC<inline-formula id="inf153">
<mml:math id="m159">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA employs a novel method for defining contact boundaries, based on the maximum distance between the alpha-carbons of amino acid pairs collected from all experimentally available proteins in the PDB.</p>
<p>Contact calculation between residues provides essential information on protein structure, function, stability, evolution, and molecular interactions. Although existing algorithms perform well for individual structures, their computational cost can become limiting in large-scale applications, such as analysis of structural databases like the PDB and AFDB, and molecular dynamics simulations where contacts must be recalculated for thousands of frames. More efficient methods, like the one we present here, enable faster processing and broader analyses across large datasets.</p>
<p>By leveraging structural and physicochemical knowledge of amino acids, we derived optimal main-chain alpha-carbon cutoff values for each amino acid pair, which significantly reduces the computational cost of detecting interatomic contacts. Advanced data structures such as <inline-formula id="inf154">
<mml:math id="m160">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-d trees are efficient for large entries but introduce unnecessary overhead for the smaller proteins that dominate the PDB. Instead, COC<inline-formula id="inf155">
<mml:math id="m161">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA employs a structure-based approach that prunes the search space based on C<inline-formula id="inf156">
<mml:math id="m162">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> distances, which undergo fewer conformational changes than side-chain atoms.</p>
<p>This approach led to a stable and efficient method tailored to real-world structural datasets. This strategic simplicity not only outperforms more complex alternatives in practice but also simplifies implementation by requiring neither external libraries nor advanced programming skills. Its scalable performance makes COC<inline-formula id="inf157">
<mml:math id="m163">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA suitable for a wide range of structural bioinformatics applications, including macromolecular interaction modeling, functional site prediction, high-throughput structural analysis, and studies of protein evolution.</p>
<p>The current version of COC<inline-formula id="inf158">
<mml:math id="m164">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA generates a &#x2018;.csv&#x2019; output file containing comprehensive information for each detected contact, including chain name, residue name and number, atom name, distance between the atom pairs, and contact type. Because the tool identifies contacts across all residues, the results can be classified as either intra-chain or inter-chain contacts, the latter being particularly valuable for analyses of protein-protein or protein-ligand interfaces. COC<inline-formula id="inf159">
<mml:math id="m165">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>DA is implemented in Python, and the full source code is publicly available at <ext-link ext-link-type="uri" xlink:href="https://github.com/LBS-UFMG/COCaDA">https://github.com/LBS-UFMG/COCaDA</ext-link>.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://github.com/lbs-ufmg/cocada_supplementary">https://github.com/lbs-ufmg/cocada_supplementary</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>RL: Writing &#x2013; review and editing, Software, Methodology, Writing &#x2013; original draft, Formal Analysis, Visualization, Data curation, Validation, Investigation, Conceptualization. DM: Validation, Visualization, Formal Analysis, Writing &#x2013; original draft, Investigation, Data curation, Methodology, Writing &#x2013; review and editing. SS: Supervision, Writing &#x2013; review and editing, Methodology, Conceptualization, Writing &#x2013; original draft. Raquel CM-M: Writing &#x2013; original draft, Methodology, Resources, Project administration, Validation, Conceptualization, Supervision, Funding acquisition, Writing &#x2013; review and editing.</p>
</sec>
<ack>
<title>Acknowledgements</title>
<p>The authors thank the funding agencies: Coordena&#xe7;&#xe3;o de Aperfei&#xe7;oamento de Pessoal de N&#xed;vel Superior (CAPES), Funda&#xe7;&#xe3;o de Amparo &#xe0; Pesquisa do Estado de Minas Gerais (FAPEMIG), and Conselho Nacional de Desenvolvimento Cient&#xed;fico e Tecnol&#xf3;gico (CNPq).</p>
</ack>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fbinf.2025.1630078/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fbinf.2025.1630078/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1076246/overview">Fabricio Martins Lopes</ext-link>, Universidade Tecnol&#xf3;gica Federal do Paran&#xe1; (UTFPR), Brazil</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1720894/overview">Yuki Kagaya</ext-link>, Purdue University, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3083967/overview">Gaihua Zhang</ext-link>, Hunan Normal University, China</p>
</fn>
<fn id="n1">
<label>1</label>
<p>Available at <ext-link ext-link-type="uri" xlink:href="https://www.rcsb.org/stats/explore/polymer_entity_type">https://www.rcsb.org/stats/explore/polymer_entity_type</ext-link>. Accessed 23 August 2024.</p>
</fn>
<fn id="n2">
<label>2</label>
<p>Available at <ext-link ext-link-type="uri" xlink:href="https://www.rcsb.org/stats/growth/growth-protein">https://www.rcsb.org/stats/growth/growth-protein</ext-link>. Accessed 23 August 2024.</p>
</fn>
<fn id="n3">
<label>3</label>
<p>Available at <ext-link ext-link-type="uri" xlink:href="https://github.com/PDBeurope/arpeggio/">https://github.com/PDBeurope/arpeggio/</ext-link>.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abramson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Adler</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dunger</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Accurate structure prediction of biomolecular interactions with AlphaFold 3</article-title>. <source>Nature</source> <volume>630</volume>, <fpage>493</fpage>&#x2013;<lpage>500</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-024-07487-w</pub-id>
<pub-id pub-id-type="pmid">38718835</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barroso</surname>
<given-names>J. R. M. S.</given-names>
</name>
<name>
<surname>Mariano</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Dias</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Rocha</surname>
<given-names>R. E. O.</given-names>
</name>
<name>
<surname>Santos</surname>
<given-names>L. H.</given-names>
</name>
<name>
<surname>Nagem</surname>
<given-names>R. A. P.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Proteus: an algorithm for proposing stabilizing mutation pairs based on interactions observed in known protein 3D structures</article-title>. <source>BMC Bioinforma.</source> <volume>21</volume>, <fpage>275</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-020-03575-6</pub-id>
<pub-id pub-id-type="pmid">32611389</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bentley</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>1975</year>). <article-title>Multidimensional binary search trees used for associative searching</article-title>. <source>Commun. ACM</source> <volume>18</volume>, <fpage>509</fpage>&#x2013;<lpage>517</lpage>. <pub-id pub-id-type="doi">10.1145/361002.361007</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Berg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cheong</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Kreveld</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Overmars</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2008</year>). &#x201c;<article-title>Orthogonal range searching</article-title>,&#x201d; in <source>Computational geometry</source> (<publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>95</fpage>&#x2013;<lpage>120</lpage>.</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berman</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Westbrook</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gilliland</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bhat</surname>
<given-names>T. N.</given-names>
</name>
<name>
<surname>Weissig</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2000</year>). <article-title>The protein data bank</article-title>. <source>Nucleic Acids Res.</source> <volume>28</volume>, <fpage>235</fpage>&#x2013;<lpage>242</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.235</pub-id>
<pub-id pub-id-type="pmid">10592235</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bickerton</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Higueruelo</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Blundell</surname>
<given-names>T. L.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Comprehensive, atomic-level characterization of structurally characterized protein-protein interactions: the PICCOLO database</article-title>. <source>BMC Bioinforma.</source> <volume>12</volume>, <fpage>313</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-12-313</pub-id>
<pub-id pub-id-type="pmid">21801404</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brown</surname>
<given-names>R. A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Building a balanced <italic>k</italic>-d tree in <italic>o(kn</italic> log <italic>n</italic>) time</article-title>. <source>J. Comput. Graph. Tech. (JCGT)</source> <volume>4</volume>, <fpage>50</fpage>&#x2013;<lpage>68</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1410.5420</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brown</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Gerlt</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Seffernick</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Babbitt</surname>
<given-names>P. C.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>A gold standard set of mechanistically diverse enzyme superfamilies</article-title>. <source>Genome Biol.</source> <volume>7</volume>, <fpage>R8</fpage>. <pub-id pub-id-type="doi">10.1186/gb-2006-7-1-r8</pub-id>
<pub-id pub-id-type="pmid">16507141</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cock</surname>
<given-names>P. J. A.</given-names>
</name>
<name>
<surname>Antao</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Chapman</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Cox</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Dalke</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Biopython: freely available python tools for computational molecular biology and bioinformatics</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>1422</fpage>&#x2013;<lpage>1423</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp163</pub-id>
<pub-id pub-id-type="pmid">19304878</pub-id>
</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>da Silveira</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Pires</surname>
<given-names>D. E. V.</given-names>
</name>
<name>
<surname>Minardi</surname>
<given-names>R. C.</given-names>
</name>
<name>
<surname>Ribeiro</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Veloso</surname>
<given-names>C. J. M.</given-names>
</name>
<name>
<surname>Lopes</surname>
<given-names>J. C. D.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Protein cutoff scanning: a comparative analysis of cutoff dependent and cutoff free methods for prospecting contacts in proteins</article-title>. <source>Proteins</source> <volume>74</volume>, <fpage>727</fpage>&#x2013;<lpage>743</lpage>. <pub-id pub-id-type="doi">10.1002/prot.22187</pub-id>
<pub-id pub-id-type="pmid">18704933</pub-id>
</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Delaunay</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>1934</year>). &#x201c;<article-title>Sur la sph&#xe8;re vide. &#xc0; la m&#xe9;moire de georges vorono&#xef;</article-title>,&#x201d; in <source>Bulletin de l&#x2019;Acad&#xe9;mie des Sciences de l&#x2019;URSS. Classe des sciences math&#xe9;matiques et naturelles VII</source>, <fpage>793</fpage>&#x2013;<lpage>800</lpage>.</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Kihara</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Computational methods for predicting protein-protein interactions using various protein features</article-title>. <source>Curr. Protoc. Protein Sci.</source> <volume>93</volume>, <fpage>e62</fpage>. <pub-id pub-id-type="doi">10.1002/cpps.62</pub-id>
<pub-id pub-id-type="pmid">29927082</pub-id>
</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dos Santos</surname>
<given-names>V. P.</given-names>
</name>
<name>
<surname>Rodrigues</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dutra</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bastos</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Mariano</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mendon&#xe7;a</surname>
<given-names>J. G.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>E-Volve: understanding the impact of mutations in SARS-CoV-2 variants spike protein on antibodies and ACE2 affinity through patterns of chemical interactions at protein interfaces</article-title>. <source>PeerJ</source> <volume>10</volume>, <fpage>e13099</fpage>. <pub-id pub-id-type="doi">10.7717/peerj.13099</pub-id>
<pub-id pub-id-type="pmid">35341044</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fassio</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Santos</surname>
<given-names>L. H.</given-names>
</name>
<name>
<surname>Silveira</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>de Melo-Minardi</surname>
<given-names>R. C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Napoli: a graph-based strategy to detect and visualize conserved protein-ligand interactions in large-scale</article-title>. <source>IEEE/ACM Trans. Comp. Biol. Bioinf.</source> <volume>17</volume>, <fpage>1317</fpage>&#x2013;<lpage>1328</lpage>. <pub-id pub-id-type="doi">10.1109/tcbb.2019.2892099</pub-id>
<pub-id pub-id-type="pmid">30629512</pub-id>
</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fassio</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Shub</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ponzoni</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>McKinley</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>O&#x2019;Meara</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>R. S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Prioritizing virtual screening with interpretable interaction fingerprints</article-title>. <source>J. Chem. Inf. Model.</source> <volume>62</volume>, <fpage>4300</fpage>&#x2013;<lpage>4318</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.2c00695</pub-id>
<pub-id pub-id-type="pmid">36102784</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Godzik</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kolinski</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Skolnick</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1992</year>). <article-title>Topology fingerprint approach to the inverse protein folding problem</article-title>. <source>J. Mol. Biol.</source> <volume>227</volume>, <fpage>227</fpage>&#x2013;<lpage>238</lpage>. <pub-id pub-id-type="doi">10.1016/0022-2836(92)90693-e</pub-id>
<pub-id pub-id-type="pmid">1522587</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jubb</surname>
<given-names>H. C.</given-names>
</name>
<name>
<surname>Higueruelo</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Ochoa-Monta&#xf1;o</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Pitt</surname>
<given-names>W. R.</given-names>
</name>
<name>
<surname>Ascher</surname>
<given-names>D. B.</given-names>
</name>
<name>
<surname>Blundell</surname>
<given-names>T. L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Arpeggio: a web server for calculating and visualising interatomic interactions in protein structures</article-title>. <source>J. Mol. Biol.</source> <volume>429</volume>, <fpage>365</fpage>&#x2013;<lpage>371</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2016.12.004</pub-id>
<pub-id pub-id-type="pmid">27964945</pub-id>
</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jumper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Figurnov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ronneberger</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Highly accurate protein structure prediction with AlphaFold</article-title>. <source>Nature</source> <volume>596</volume>, <fpage>583</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id>
<pub-id pub-id-type="pmid">34265844</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kendrew</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Bodo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dintzis</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Parrish</surname>
<given-names>R. G.</given-names>
</name>
<name>
<surname>Wyckoff</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Phillips</surname>
<given-names>D. C.</given-names>
</name>
</person-group> (<year>1958</year>). <article-title>A three-dimensional model of the myoglobin molecule obtained by x-ray analysis</article-title>. <source>Nature</source> <volume>181</volume>, <fpage>662</fpage>&#x2013;<lpage>666</lpage>. <pub-id pub-id-type="doi">10.1038/181662a0</pub-id>
<pub-id pub-id-type="pmid">13517261</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kovalevskiy</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Mateos-Garcia</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tunyasuvunakool</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>AlphaFold two years on: validation and impact</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>121</volume>, <fpage>e2315002121</fpage>. <pub-id pub-id-type="doi">10.1073/pnas.2315002121</pub-id>
<pub-id pub-id-type="pmid">39133843</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laskowski</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Swindells</surname>
<given-names>M. B.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>LigPlot&#x2b;: multiple ligand-protein interaction diagrams for drug discovery</article-title>. <source>J. Chem. Inf. Model.</source> <volume>51</volume>, <fpage>2778</fpage>&#x2013;<lpage>2786</lpage>. <pub-id pub-id-type="doi">10.1021/ci200227u</pub-id>
<pub-id pub-id-type="pmid">21919503</pub-id>
</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mancini</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Higa</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Oliveira</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dominiquini</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kuser</surname>
<given-names>P. R.</given-names>
</name>
<name>
<surname>Yamagishi</surname>
<given-names>M. E. B.</given-names>
</name>
<etal/>
</person-group> (<year>2004</year>). <article-title>STING contacts: a web-based application for identification and analysis of amino acid contacts within protein structure and across protein interfaces</article-title>. <source>Bioinformatics</source> <volume>20</volume>, <fpage>2145</fpage>&#x2013;<lpage>2147</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bth203</pub-id>
<pub-id pub-id-type="pmid">15073001</pub-id>
</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mariano</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Da Fonseca J&#xfa;nior</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Santos</surname>
<given-names>L. H.</given-names>
</name>
<name>
<surname>de Melo-Minardi</surname>
<given-names>R. C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Editorial: bioinformatics in the age of data science: algorithms, methods, and tools applied from omics to structural data</article-title>. <source>Front. Bioinforma.</source> <volume>3</volume>, <fpage>1246859</fpage>. <pub-id pub-id-type="doi">10.3389/fbinf.2023.1246859</pub-id>
<pub-id pub-id-type="pmid">37469552</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mura</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Draizen</surname>
<given-names>E. J.</given-names>
</name>
<name>
<surname>Bourne</surname>
<given-names>P. E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Structural biology meets data science: does anything change?</article-title> <source>Curr. Opin. Struct. Biol.</source> <volume>52</volume>, <fpage>95</fpage>&#x2013;<lpage>102</lpage>. <pub-id pub-id-type="doi">10.1016/j.sbi.2018.09.003</pub-id>
<pub-id pub-id-type="pmid">30267935</pub-id>
</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mondal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Khatua</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ghosh</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Big data in biology: the hope and present-day challenges in it</article-title>. <source>Gene Rep.</source> <volume>21</volume>, <fpage>100869</fpage>. <pub-id pub-id-type="doi">10.1016/j.genrep.2020.100869</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pimentel</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Mariano</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Cant&#xe3;o</surname>
<given-names>L. X. S.</given-names>
</name>
<name>
<surname>Bastos</surname>
<given-names>L. L.</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>de Lima</surname>
<given-names>L. H. F.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>VTR: a web tool for identifying analogous contacts on protein structures and their complexes</article-title>. <source>F. Bioinf.</source> <volume>1</volume>, <fpage>730350</fpage>. <pub-id pub-id-type="doi">10.3389/fbinf.2021.730350</pub-id>
<pub-id pub-id-type="pmid">36303745</pub-id>
</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pires</surname>
<given-names>D. E. V.</given-names>
</name>
<name>
<surname>de Melo-Minardi</surname>
<given-names>R. C.</given-names>
</name>
<name>
<surname>dos Santos</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>da Silveira</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Santoro</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Meira</surname>
<given-names>W.</given-names>
<suffix>Jr</suffix>
</name>
</person-group> (<year>2011</year>). <article-title>Cutoff scanning matrix: structural classification and function prediction by protein inter-residue distance patterns</article-title>. <source>BMC Gen.</source> <volume>12</volume>, <fpage>S12</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-12-S4-S12</pub-id>
</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schreyer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Blundell</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>CREDO: a protein-ligand interaction database for drug discovery</article-title>. <source>Chem. Biol. Drug Des.</source> <volume>73</volume>, <fpage>157</fpage>&#x2013;<lpage>167</lpage>. <pub-id pub-id-type="doi">10.1111/j.1747-0285.2008.00762.x</pub-id>
<pub-id pub-id-type="pmid">19207418</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schreyer</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Blundell</surname>
<given-names>T. L.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>CREDO: a structural interactomics database for drug discovery</article-title>. <source>Database (Oxford)</source> <volume>2013</volume>, <fpage>bat049</fpage>. <pub-id pub-id-type="doi">10.1093/database/bat049</pub-id>
<pub-id pub-id-type="pmid">23868908</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Silva</surname>
<given-names>M. F. M.</given-names>
</name>
<name>
<surname>Martins</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Mariano</surname>
<given-names>D. C. B.</given-names>
</name>
<name>
<surname>Santos</surname>
<given-names>L. H.</given-names>
</name>
<name>
<surname>Pastorini</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Pantuza</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Proteingo: motivation, user experience, and learning of molecular interactions in biological complexes</article-title>. <source>Entertain. Comput.</source> <volume>29</volume>, <fpage>31</fpage>&#x2013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1016/j.entcom.2018.11.001</pub-id>
</mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Smetana</surname>
<given-names>J. H. C.</given-names>
</name>
<name>
<surname>Misra</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Principles of protein structure and function</article-title>,&#x201d; in <source>Intro. to biomol. Struct. and biophys.</source> (<publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>32</lpage>.</mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sobolev</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Sorokine</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Prilusky</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Abola</surname>
<given-names>E. E.</given-names>
</name>
<name>
<surname>Edelman</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Automated analysis of interatomic contacts in proteins</article-title>. <source>Bioinformatics</source> <volume>15</volume>, <fpage>327</fpage>&#x2013;<lpage>332</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/15.4.327</pub-id>
<pub-id pub-id-type="pmid">10320401</pub-id>
</mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vangone</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bonvin</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Contacts-based prediction of binding affinity in protein-protein complexes</article-title>. <source>Elife</source> <volume>4</volume>, <fpage>e07454</fpage>. <pub-id pub-id-type="doi">10.7554/elife.07454</pub-id>
<pub-id pub-id-type="pmid">26193119</pub-id>
</mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Varadi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bertoni</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Magana</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Paramval</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Pidruchna</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Radhakrishnan</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>AlphaFold protein structure database in 2024: providing structure coverage for over 214 million protein sequences</article-title>. <source>Nucleic Acids Res.</source> <volume>52</volume>, <fpage>D368</fpage>&#x2013;<lpage>D375</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkad1011</pub-id>
<pub-id pub-id-type="pmid">37933859</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Voronoi</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1908</year>). <article-title>Nouvelles applications des param&#xe8;tres continus &#xe0; la th&#xe9;orie des formes quadratiques. deuxi&#xe8;me m&#xe9;moire. recherches sur les parall&#xe9;llo&#xe8;dres primitifs</article-title>. <source>J. f&#xfc;r die reine und angewandte Math.</source> <volume>134</volume>, <fpage>198</fpage>&#x2013;<lpage>287</lpage>. <pub-id pub-id-type="doi">10.1515/crll.1908.133.97</pub-id>
</mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wald</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Havran</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>On building fast kd-trees for ray tracing, and on doing that in O(N log n)</article-title>,&#x201d; in <source>2006 IEEE symposium on interactive ray tracing</source>.</mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wallace</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Laskowski</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Thornton</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>LIGPLOT: a program to generate schematic diagrams of protein-ligand interactions</article-title>. <source>Protein Eng. Des. Sel.</source> <volume>8</volume>, <fpage>127</fpage>&#x2013;<lpage>134</lpage>. <pub-id pub-id-type="doi">10.1093/protein/8.2.127</pub-id>
<pub-id pub-id-type="pmid">7630882</pub-id>
</mixed-citation>
</ref>
</ref-list>
</back>
</article>
