<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2024.1352703</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Investigating the ability of deep learning-based structure prediction to extrapolate and/or enrich the set of antibody CDR canonical forms</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Greenshields-Watson</surname>
<given-names>Alexander</given-names>
</name>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2341915"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Abanades</surname>
<given-names>Brennan</given-names>
</name>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2139894"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Deane</surname>
<given-names>Charlotte M.</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/584257"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<institution>Oxford Protein Informatics Group, Department of Statistics, University of Oxford</institution>, <addr-line>Oxford</addr-line>, <country>United Kingdom</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Musie Ghebremichael, Harvard University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Sai Pooja Mahajan, Johns Hopkins University, United States</p>
<p>Traian Sulea, National Research Council Canada (NRC), Canada</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Charlotte M. Deane, <email xlink:href="mailto:deane@stats.ox.ac.uk">deane@stats.ox.ac.uk</email>
</p>
</fn>
<fn fn-type="other" id="fn003">
<p>&#x2020;ORCID: Alexander Greenshields-Watson, <uri xlink:href="https://orcid.org/0000-0002-8740-9823">orcid.org/0000-0002-8740-9823</uri>; Brennan Abanades, <uri xlink:href="https://orcid.org/0000-0001-8712-533X">orcid.org/0000-0001-8712-533X</uri>; Charlotte M. Deane, <uri xlink:href="https://orcid.org/0000-0003-1388-2252">orcid.org/0000-0003-1388-2252</uri>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>28</day>
<month>02</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1352703</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>01</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Greenshields-Watson, Abanades and Deane</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Greenshields-Watson, Abanades and Deane</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Deep learning models have been shown to accurately predict protein structure from sequence, allowing researchers to explore protein space from the structural viewpoint. In this paper we explore whether &#x201c;novel&#x201d; features, such as distinct loop conformations can arise from these predictions despite not being present in the training data. Here we have used ABodyBuilder2, a deep learning antibody structure predictor, to predict the structures of ~1.5M paired antibody sequences. We examined the predicted structures of the canonical CDR loops and found that most of these predictions fall into the already described CDR canonical form structural space. We also found a small number of &#x201c;new&#x201d; canonical clusters composed of heterogeneous sequences united by a common sequence motif and loop conformation. Analysis of these novel clusters showed their origins to be either shapes seen in the training data at very low frequency or shapes seen at high frequency but at a shorter sequence length. To evaluate explicitly the ability of ABodyBuilder2 to extrapolate, we retrained several models whilst withholding all antibody structures of a specific CDR loop length or canonical form. These &#x201c;starved&#x201d; models showed evidence of generalisation across CDRs of different lengths, but they did not extrapolate to loop conformations which were highly distinct from those present in the training data. However, the models were able to accurately predict a canonical form even if only a very small number of examples of that shape were in the training data. Our results suggest that deep learning protein structure prediction methods are unable to make completely out-of-domain predictions for CDR loops. However, in our analysis we also found that even minimal amounts of data of a structural shape allow the method to recover its original predictive abilities. We have made the ~1.5 M predicted structures used in this study available to download at <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.10280181">https://doi.org/10.5281/zenodo.10280181</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>antibody</kwd>
<kwd>canonical forms</kwd>
<kwd>structure prediction</kwd>
<kwd>complementarity determining regions</kwd>
<kwd>deep learning</kwd>
</kwd-group>
<contract-num rid="cn001">EP/S024093/1</contract-num>
<contract-sponsor id="cn001">Engineering and Physical Sciences Research Council<named-content content-type="fundref-id">10.13039/501100000266</named-content>
</contract-sponsor>
<counts>
<fig-count count="7"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="48"/>
<page-count count="19"/>
<word-count count="11345"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Systems Immunology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>Deep learning has revolutionised the field of structural biology with tools such as AlphaFold2 (AF2) (<xref ref-type="bibr" rid="B1">1</xref>), RosettaFold (<xref ref-type="bibr" rid="B2">2</xref>) and ESMFold (<xref ref-type="bibr" rid="B3">3</xref>) that can accurately predict protein tertiary structure from primary sequence. These tools are all trained on the known protein structure landscape derived from the PDB (<xref ref-type="bibr" rid="B4">4</xref>) and have been shown to generalise well to proteins that were not seen during training. Several studies have used these models to enrich the existing protein structure landscape by making extensive predictions from the larger available sequence space. Analysis of these predictions revealed many examples of structures that are very different from the closest available match in experimentally defined data (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B5">5</xref>).</p>
<p>By analysing over 365,000 high confidence structures predicted by AF2, Bordin et&#xa0;al. were able to define 25 novel superfamilies which did not cluster into any existing CATH classifications using their CATH-Assign protocol (<xref ref-type="bibr" rid="B5">5</xref>). A second example of new knowledge arising from structural predictions was provided by ESMFold (<xref ref-type="bibr" rid="B3">3</xref>). Here, Lin et&#xa0;al. predicted the structures of over 600M metagenomic sequences isolated from diverse environmental and clinical samples. The use of these metagenomic sequences increased the probability of finding examples that were highly distant from the sequence and structural data used to train ESM2 and ESMFold respectively (<xref ref-type="bibr" rid="B3">3</xref>). Within a sample of 1M modelled structures defined as high confidence (predicted local distance difference test score, pLDDT&#xa0;&gt;&#xa0;0.7 and predicted template modelling score, pTM &gt; 0.7), the authors found over 125,000 predictions with no close match in the PDB [defined as pTM &gt; 0.5 carried out using Foldseek (<xref ref-type="bibr" rid="B6">6</xref>)] and in close alignment to the corresponding predictions from AF2. While both studies demonstrate that structure prediction tools can confidently generate novel structures, X-ray crystallography data was not obtained to conclusively validate the predictions. It is also not clear if the novel structures generated are composites of large substructural fragments present in the training data.</p>
<p>To attempt to explicitly address whether models can generalise to unseen regions of structural space, Ahdritz et&#xa0;al. carried out &#x2018;out-of-domain&#x2019; experiments using OpenFold (<xref ref-type="bibr" rid="B7">7</xref>). In particular, examining if OpenFold can generalise from limited data to accurately predict alpha helices or beta sheets despite their omission from training datasets. However, they were not able to completely remove all signal of these secondary structures from their training data, and hence the models were likely still learning from a much-reduced set of examples, rather than extrapolating to a completely unknown structure based on their induction of biophysical rules.</p>
<p>These analyses raise the question of whether current deep learning-based models are truly capable of predicting conformations which are never present in training data. While extrapolation by deep neural networks is theoretically plausible (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B9">9</xref>) searching for evidence of this is difficult and requires extensive classification of training data and the resulting predictions.</p>
<p>One limitation of deep learning based protein structure predictors is their poor performance on stretches of sequence that are intrinsically disordered (<xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B11">11</xref>) or explore diverse conformational space (<xref ref-type="bibr" rid="B12">12</xref>). The loops of adaptive immune receptors, antibodies, and T cell receptors, fall into the latter category. These loops form the majority of the binding site (paratope) of these proteins and are termed complementarity determining regions (CDRs) (<xref ref-type="bibr" rid="B13">13</xref>).</p>
<p>The protein sequences that make up the CDR loops arise from two genetic mechanisms, termed V(D)J recombination (<xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B15">15</xref>) and somatic hypermutation (<xref ref-type="bibr" rid="B16">16</xref>). In antibodies, the process of V(D)J recombination randomly pairs V-, D- and J-genes (VJ genes for light chains and VDJ genes for heavy chains) and introduces junctional diversity through insertion and deletion of nucleotides. Further diversity is introduced to the V-gene region of the antibody by somatic hypermutation, where point mutations that modify the amino acid sequence and improve binding affinity are positively selected and progressively dominate the immune response to a pathogen. These mechanisms create high levels of sequence diversity and are evolutionarily advantageous as they combine to provide a nearly limitless potential of binding solutions which allow antibodies to neutralise the correspondingly limitless diversity of pathogens to which humans can be exposed (<xref ref-type="bibr" rid="B17">17</xref>). Of the six CDR loops on an antibody, diversity is highest within the CDRH3, the residues of which often disproportionally govern paratope-epitope interactions (<xref ref-type="bibr" rid="B18">18</xref>). Structure predictors have been found to perform poorly on this region of antibodies, for example with average RMSD values between predictions and ground truth that exceed 2.5 &#xc5; for state-of-the-art models such as AlphaFold-Multimer (AFM) (<xref ref-type="bibr" rid="B19">19</xref>) and ABodyBuilder2 (ABB2) (<xref ref-type="bibr" rid="B20">20</xref>).</p>
<p>The predictive performance on the remaining five loops, CDRL1-3 and CDRH1-2, is far better (average RMSD &lt;1&#xc5;), despite these being subject to the genetic process of somatic hypermutation and being influenced by neighbouring hypervariable loops (<xref ref-type="bibr" rid="B21">21</xref>). The ability to accurately model these can be explained by canonical forms, the term given to sets of CDR loops of the same length that adopt similar backbone conformations and share a sequence motif (<xref ref-type="bibr" rid="B22">22</xref>). These canonical forms were first observed in crystallographic datasets of available antibody structures before 1986 (<xref ref-type="bibr" rid="B23">23</xref>). With the deposition of more structural data both the number of canonical conformations and the sequences that could be assigned to each were continuously expanded and redefined (<xref ref-type="bibr" rid="B24">24</xref>&#x2013;<xref ref-type="bibr" rid="B29">29</xref>). This information linked diverse sets of sequences to distinct loop conformations and thus was increasingly useful to antibody researchers, by providing a form of sequence-to-structure prediction that could be automated by template search and homology modelling tools (<xref ref-type="bibr" rid="B27">27</xref>, <xref ref-type="bibr" rid="B30">30</xref>).</p>
<p>The latest CDR structure and sequence pairings harvested from antibody structural data are defined in PyIgClassify2 (<xref ref-type="bibr" rid="B28">28</xref>). The definitions were released as the &#x2018;penultimate classification of canonical forms&#x2019; in reference to the breakthroughs in structure prediction research that may soon render the predictive power of this relationship obsolete. While structure prediction methods are still being evaluated, especially in the domain of adaptive immune receptors, PyIgClassify2 can serve as a map of the known conformational space explored by antibody CDRs.</p>
<p>Using the rigorous definitions from PyIgClassify2 as a reference point in structural space, we set out to test whether the predicted structures from all available paired antibody sequences in observed antibody space, OAS, (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B32">32</xref>) reveal novel canonical clusters or highlight conformations not explored by existing experimental data. We then assess these new areas to determine whether they represent evidence of extrapolation or had direct origins in the training data.</p>
<p>ABodyBuilder2 (ABB2) is a structure prediction tool specific to antibodies (<xref ref-type="bibr" rid="B20">20</xref>). It uses an ensemble of four deep learning models trained on the structures of over 3500 antibodies as well as a fast minimisation in the AMBER14 forcefield (<xref ref-type="bibr" rid="B33">33</xref>, <xref ref-type="bibr" rid="B34">34</xref>) to make predictions with comparable accuracy to AFM in a fraction of the time. We used ABB2 to predict the structures of ~1.5M paired antibody sequences. We mapped the conformational space of the CDRL1-3 and CDRH1-2 loops and used existing classifications of the canonical forms in experimental data as reference points. By comparing the loop conformations of canonical clusters to clusters found in predictions derived from the ~1.5M heterogeneous sequences we were able to redefine and identify new canonical clusters.</p>
<p>These new clusters (potential canonical forms) were defined by unique sequence motifs and shared loop conformations and typically arose from enrichment of a small number of examples in the experimental data. We also observed apparently novel clusters (canonical forms) that derived from similar shapes (and sequences) of a different loop length, a phenomenon that has been previously described within the structural dataset (<xref ref-type="bibr" rid="B26">26</xref>), termed length independence.</p>
<p>Using our mapping of structural space and the definitions of canonical forms we designed out-of-domain retraining experiments which explicitly tested the capability of ABB2 to both generalise and extrapolate. We found that with zero examples of a given CDR shape ABB2 was consistently unable to predict it. However, with the introduction of very small numbers of a shape, the predictive ability was restored. Overall, these analyses exemplify the power of augmenting experimental data with predictions and provide simple tests for extrapolation and effective data recapitulation that may help inform the next generation of structure predictors.</p>
</sec>
<sec id="s2">
<title>Methods</title>
<sec id="s2_1">
<title>Selection of paired antibodies sequences for ABodyBuilder2</title>
<p>Paired antibody sequences were retrieved from observed antibody space (OAS) (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B32">32</xref>) (1-March-2023, <ext-link ext-link-type="uri" xlink:href="https://opig.stats.ox.ac.uk/webapps/oas/">https://opig.stats.ox.ac.uk/webapps/oas/</ext-link>). Non-redundant sequence pairs were filtered by minimum lengths defined in the ABB2 workflow such as minimum residue length of 70, starting IMGT residue number less than 8, and end residue number greater than 120, (see sequence checks <ext-link ext-link-type="uri" xlink:href="https://github.com/oxpig/ImmuneBuilder">https://github.com/oxpig/ImmuneBuilder</ext-link>). Sequences containing any gaps or ambiguities were removed to leave 1,492,044 pairs for processing. ABB2 was run on all sequences and 1,492,031 structures were successfully predicted.</p>
</sec>
<sec id="s2_2">
<title>Annotation of CDR loops in paired antibody sequences</title>
<p>To group the relevant modelled structures for conformational analyses, the CDR loops of all input sequences were annotated with information on their sequence composition and length. The CDRs were defined according to the IMGT numbering scheme (CDR1: 27-38, CDR2: 56-65, CDR3: 105-117) (<xref ref-type="bibr" rid="B35">35</xref>). This numbering system was chosen as the anchor residues and CDR locations are consistent for both heavy and light chains, as well as being the standard reference point for V(D)J gene annotation. Each sequence was IMGT numbered using ANARCI (<xref ref-type="bibr" rid="B36">36</xref>) and the CDR sequences checked against corresponding information from IgBLAST annotations (<xref ref-type="bibr" rid="B37">37</xref>). For a given heavy or light chain, if there was any discrepancy between IgBLAST and ANARCI CDR definitions then this chain was not taken forward for conformational analysis (resulting in the exclusion of 6414 light chains and 9822 heavy chains from the structures predicted above).</p>
</sec>
<sec id="s2_3">
<title>Retrieval and selection of experimental structures from SAbDab</title>
<p>The structural antibody database (SAbDab) (<xref ref-type="bibr" rid="B38">38</xref>, <xref ref-type="bibr" rid="B39">39</xref>) is a curated database which contains all antibody, single chain variable fragment (SCFV) and nanobody structures available in the PDB (<xref ref-type="bibr" rid="B4">4</xref>). IMGT numbered structures used in ABB2 test, train and validation datasets were downloaded from SAbDab. These structures were derived by X-ray crystallography or cryogenic electron microscopy (cryo-EM) and with a resolution better than 3.5 &#xc5; (full list given in SI of <xref ref-type="bibr" rid="B20">20</xref>).</p>
</sec>
<sec id="s2_4">
<title>Annotation of SAbDab structures with PyIgClassify2 information</title>
<p>The information on CDR loop canonical forms was obtained from the pyig_cdr_data.txt file downloaded from the PyIgClassify2 website (<ext-link ext-link-type="uri" xlink:href="http://dunbrack2.fccc.edu/PyIgClassify2/">http://dunbrack2.fccc.edu/PyIgClassify2/</ext-link>, 21-Feb-2023). This provides complete information for all CDR loops on each structure within a given PDB file, including information on sequence identical but structurally distinct members of the asymmetric unit which are distinguished by their PDB chain identifier. For each loop the relevant information includes the length, sequence, canonical form assignment (both with and without an electron density confidence cut off), PDB identifier of the parent chain and information on whether the CDR is structurally complete or is missing any backbone coordinates. This data was used to filter the relevant structures and their CDR loops for each analysis, and to annotate the experimental data points by canonical cluster membership.</p>
</sec>
<sec id="s2_5">
<title>Alignment of IMGT and AHo numbering systems</title>
<p>PyIgClassify2 CDR lengths are defined according to the AHo numbering scheme (<xref ref-type="bibr" rid="B40">40</xref>) which symmetrically places insertions and deletions around positions defined as key residues in each CDR. This deviates from the IMGT numbering scheme (<xref ref-type="bibr" rid="B35">35</xref>) which places insertions centrally within each CDR at fixed positions. The different approaches to defining the CDRs mean that IMGT defined lengths for CDRL1-2 and CDRH1-2 are shorter than those defined in the AHo numbering scheme used in PyIgClassify2. Therefore, for each CDR and length combination analysed in this study, we have listed the corresponding AHo CDR lengths in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, along with the PyIgClassify2 defined canonical forms and sequence motifs described in (<xref ref-type="bibr" rid="B28">28</xref>).</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Alignment of IGMT CDR numbering, AHo CDR numbering and PyIgClassify2 Canonical Forms.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">CDR</th>
<th valign="top" align="left">IMGT Length</th>
<th valign="top" align="left">Aho Length</th>
<th valign="top" colspan="2" align="left">PyIgClassify2 Defined Canonical Forms (sequence motifs)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">L1</td>
<td valign="top" align="left">6</td>
<td valign="top" align="left">11</td>
<td valign="top" align="left">L1-11-1 (RASQsISsyLA)<break/>L1-11-2 (RASQDIsnYLA)</td>
<td valign="top" align="left">L1-11-3 (gGDniGDKsVH)<break/>L1-11-4 (SGDaLpKKYAY)</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">7</td>
<td valign="top" align="left">12</td>
<td valign="top" align="left">L1-12-1 (RASqSVSSSYLa)<break/>L1-12-2 (RASQSVSSNYLA)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">8</td>
<td valign="top" align="left">13</td>
<td valign="top" align="left">L1-13-1 (SGSSSNIGsNYVS)<break/>L1-13-2 (TRSSGsIaSNYVq)</td>
<td valign="top" align="left">L1-13-3 (QSSQSVYNNNNLA)</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">9</td>
<td valign="top" align="left">14</td>
<td valign="top" align="left">L1-14-1 (RSStGAVTtSNyAN)<break/>L1-14-2 (TGTSSDvGgYNYVS)</td>
<td valign="top" align="left">L1-14-3 (TGSSSNIGAGYDVH)</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">11</td>
<td valign="top" align="left">16</td>
<td valign="top" align="left">L1-16-1 (RSSQSLVHSNGNTYLe)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">12</td>
<td valign="top" align="left">17</td>
<td valign="top" align="left">L1-17-1 (KSSQSLLySSNqKNYLA)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">L2</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">8</td>
<td valign="top" align="left">L2-8-1 (YdaSnrAS)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">L3</td>
<td valign="top" align="left">8</td>
<td valign="top" align="left">8</td>
<td valign="top" align="left">L3-8-1 (qQYyNlWT)<break/>L3-8-3 (QQYYSSPT)</td>
<td valign="top" align="left">L3-8-4 (QQYdssPT)</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">9</td>
<td valign="top" align="left">9</td>
<td valign="top" align="left">L3-9-1 (QqWDSshwv)<break/>L3-9-2 (QQyystPYT)<break/>L3-9-3 (QsydsSsvv)</td>
<td valign="top" align="left">L3-9-4 (ALWYSsHWV)<break/>L3-9-cis7-1 (QQyYsYPyT)<break/>L3-9-cis7-2 (QHFWgTPRT)</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">10</td>
<td valign="top" align="left">10</td>
<td valign="top" align="left">L3-10-1 (sSYtSSsTwV)<break/>L3-10-2 (cSYAGSstwV)<break/>L3-10-3 (QvWDSssdVV)</td>
<td valign="top" align="left">L3-10-cis78-1 (qQrTHwPPLT)</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">11</td>
<td valign="top" align="left">11</td>
<td valign="top" align="left">L3-11-1 (QaWDSSlsgvV)<break/>L3-11-2 (QStDSSGTYwV)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">H1</td>
<td valign="top" align="left">8</td>
<td valign="top" align="left">13</td>
<td valign="top" align="left">H1-13-1 (aASGfTFssYwmH)<break/>H1-13-3 (aASGRTFSSYaMG)</td>
<td valign="top" align="left">H1-13-4 (aaSGGtFsgYYWS)<break/>H1-13-5 (AASGRTFSIYaMG)</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">9</td>
<td valign="top" align="left">14</td>
<td valign="top" align="left">H1-14-1 (TVtGYSITSdYaWN)<break/>H1-14-2 (AVSGGSISssYyWS)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">10</td>
<td valign="top" align="left">15</td>
<td valign="top" align="left">H1-15-1 (tFSGFSLSTSGMGVG)<break/>H1-15-2 (tvSGDSiSssdyyWg)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">H2</td>
<td valign="top" align="left">7</td>
<td valign="top" align="left">9</td>
<td valign="top" align="left">H2-9-1 (YIYYSGSTY)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">8</td>
<td valign="top" align="left">10</td>
<td valign="top" align="left">H2-10-1 (wInPgNGdTN)<break/>H2-10-2 (AISsdGssTY)<break/>H2-10-3 (EIyPGsGSTn)</td>
<td valign="top" align="left">H2-10-4 (gISSGGgYty)<break/>H2-10-6 (WINPsGGsTy)</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">10</td>
<td valign="top" align="left">11</td>
<td valign="top" align="left">H2-12-1 (RTYYRSKWYNd)</td>
<td valign="top" align="left"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>For each CDR, the IMGT lengths used in these analyses are presented alongside the corresponding AHo lengths and PyIgClassify2 defined canonical forms for that CDR. PyIgClassify2 canonical forms are named according to the CDR (e.g. L1) followed by the Aho length (e.g. -6), and the form number (e.g. -1) to form a unique identifier for each form (e.g. L1-11-6). For each canonical form the consensus sequence motif, as defined in (<xref ref-type="bibr" rid="B28">28</xref>) is given in brackets. Uppercase letters of the consensus sequence indicate highly conserved amino acids at a given position, while lowercase letters indicate a less conserved amino acid that was still observed in the cluster of loops found in the PyIgClassify2 analyses. Information is only provided for the loops analysed in this study, i.e., those which corresponded to the most dominant non-redundant sequences in OAS, (see <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>).</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s2_6">
<title>Pre-processing of SAbDab datasets</title>
<p>SAbDab files included SCFV structures where the heavy and light chain are part of a single continuous sequence, as well as datasets with multiple sequence-identical copies in the asymmetric unit. To ensure all CDR loops could be correctly identified and consistently aligned, the relevant chains in each dataset were isolated, IMGT numbered and then saved as individual files linked to the corresponding PyIgClassify2 meta data. For SCFV structures the continuous sequence was broken and each fragment treated as an individual chain. If a chain within a dataset could not be numbered with ANARCI, or a specific CDR loop was missing residues (as indicated by the PyIgClassify2 &#x2018;cdr_ordered&#x2019; flag), they were not included in structural analyses. This processing resulted in 11821 heavy and light chains from 3355 PDB files which were used for further analyses. As ABB2 predicted structures were all correctly numbered and contained only a single copy in the asymmetric unit, they did not require any pre-processing for downstream analyses.</p>
</sec>
<sec id="s2_7">
<title>Structural analysis of CDR loops of the same length</title>
<p>To perform structural analysis, CDR loops of predicted structures were grouped according to their CDR type (i.e., CDRL1 or CDRH2), amino acid length and sequence composition (non-redundant sequences only). These were analysed alongside all relevant loop structures from SAbDab, for these experimental data points redundant sequences were included as they may contain alternate conformations of the same sequence.</p>
<p>To provide a consistent frame of reference for each CDR length, a loop template was chosen from the highest resolution PDB structure available, this structure also had to be classified as representative of a PyIgClassify2 defined canonical form (&#x2018;is_representative&#x2019; flag) and thus was not likely to be an outlier or exhibit any structural features that set it apart. All CDR loops in the analysis were aligned to this template by superimposition of the alpha carbon atoms of the 10 framework residues either side of the loop (CDR1: 22-26 &amp; 39-43, CDR2: 51-55 &amp; 66-70, CDR3: 100-104 &amp; 118-122). If superimposition resulted in RMSD values greater than 1.5 &#xc5; then these were not taken forward for loop comparisons. For predicted structures, the framework regions were highly consistent and less than 5 loops per analysis were eliminated. For experimental data points the number eliminated due to framework misalignments ranged from 0 to a maximum of 31 for CDRL1-Len-6 (out of 2499 chains), with a median number of 3 data points eliminated across all analyses.</p>
<p>The carbon and nitrogen backbone atom coordinates of the aligned loops were extracted and saved (CDR1: 27-38, CDR2: 56-65, CDR3: 105-117). Atom counts were checked and then all pairwise RMSD values calculated. This resulted in an N-by-N pairwise distance matrix of RMSD values including both predicted and experimental datapoints for each CDR loop type at every length. To limit the size of pairwise matrices, loops of predicted structures were analysed in batches of 42,000. Where a CDR and length had more than this number of non-redundant sequences (see <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>) subsequent batches of a maximum size of 42,000 were run until all relevant loop structures were analysed. All batches were run through the clustering and visualisation pipeline (see sections below) and then inspected to ensure results were consistent across all analyses. For multi-batch CDRs, graphs of the first batch are shown in main figures and graphs of subsequent batches are provided in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Information</bold>
</xref> (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;7</bold>
</xref>).</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Results of RMSD and DBS cluster analysis in high frequency CDR lengths.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">CDR</th>
<th valign="top" align="left">IMGT Length</th>
<th valign="top" align="left">Number of Unique Sequences in OAS</th>
<th valign="top" align="left">New information arising from predictions?</th>
<th valign="top" align="left">Origin of novel canonical cluster?</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">L1</td>
<td valign="top" align="left">6</td>
<td valign="top" align="left">21,707</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">N/A</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">7</td>
<td valign="top" align="left">10,212</td>
<td valign="top" align="left">New canonical cluster</td>
<td valign="top" align="left">Extra density matching existing unassigned exp data</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">8</td>
<td valign="top" align="left">7,596</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">N/A</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">9</td>
<td valign="top" align="left">13,846</td>
<td valign="top" align="left">New canonical cluster</td>
<td valign="top" align="left">Extra density matching existing unassigned exp data</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">11</td>
<td valign="top" align="left">6,992</td>
<td valign="top" align="left">Sub-division of existing cluster</td>
<td valign="top" align="left">Uneven distribution of density within existing form</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">12</td>
<td valign="top" align="left">10,565</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">N/A</td>
</tr>
<tr>
<td valign="top" align="left">L2</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">2,280</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">N/A</td>
</tr>
<tr>
<td valign="top" align="left">L3</td>
<td valign="top" align="left">8</td>
<td valign="top" align="left">15,805</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">N/A</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">9</td>
<td valign="top" align="left">76,087</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">N/A</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">10</td>
<td valign="top" align="left">56,599</td>
<td valign="top" align="left">New canonical cluster</td>
<td valign="top" align="left">Density derived from length- independent conformation and existing data</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">11</td>
<td valign="top" align="left">51,277</td>
<td valign="top" align="left">New canonical cluster</td>
<td valign="top" align="left">Density derived from length- independent conformation</td>
</tr>
<tr>
<td valign="top" align="left">H1</td>
<td valign="top" align="left">8</td>
<td valign="top" align="left">61,617</td>
<td valign="top" align="left">Sub-division of existing cluster</td>
<td valign="top" align="left">Uneven distribution of density within existing form</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">9</td>
<td valign="top" align="left">5,655</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">N/A</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">10</td>
<td valign="top" align="left">25,000</td>
<td valign="top" align="left">Sub-division of existing cluster</td>
<td valign="top" align="left">Extra density matching existing unassigned exp data</td>
</tr>
<tr>
<td valign="top" align="left">H2</td>
<td valign="top" align="left">7</td>
<td valign="top" align="left">27,769</td>
<td valign="top" align="left">Sub-division of existing cluster</td>
<td valign="top" align="left">Uneven distribution of density within existing form</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">8</td>
<td valign="top" align="left">99,003</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">N/A</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">10</td>
<td valign="top" align="left">13,114</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">N/A</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>For each CDR and IMGT length explored in this study, details of the number of non-redundant sequences (and hence loops structurally analysed) and the corresponding results of structural analyses are given. The new information arising from analysis of the predicted structures could be defined as either the identification of a new canonical cluster, or the sub-division of an existing cluster of loops that had previously been defined as belonging to a single canonical form. Further details are given on whether these clusters arose from length independent conformations and/or existing experimental data points classified as unassigned by PyIgClassify2 (<xref ref-type="bibr" rid="B28">28</xref>).</p>
<p>N/A, not applicable.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s2_8">
<title>Analysis of CDRH3 loops</title>
<p>We do not present CDRH3 analyses due to the comparatively poor prediction accuracy in this region by ABB2, and other tools such as AFM (<xref ref-type="bibr" rid="B20">20</xref>). This uncertainty meant had we found &#x201c;novel&#x201d; canonical forms or observations in the CDRH3 region, we could not have confidence that they reflected real loop conformations.</p>
<p>Furthermore, when we performed CDRH3 clustering on the high frequency shorter sequences (CDRH3 lengths 12, 13, 14, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;8</bold>
</xref>), there was a lack of high-density clusters. This was likely related to the more evenly distributed occupancy of structural space as seen from the multidimensional scaling plots. When density-based clusters were found, the logo plots were uninformative with no apparent motif present in the middle of the CDRH3, and enrichments localised to the beginning and end of the loops (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;8</bold>
</xref>). Given the heterogeneity of CDRH3 we felt these results were to be expected and this region did not warrant further exploration for novel canonical forms.</p>
</sec>
<sec id="s2_9">
<title>Structural analysis of CDR loops of different length</title>
<p>For later analyses aimed at discovering length independent conformations, we also calculated the distance between loops of different lengths. The normalised dynamic time warping (DTW) scores were used to quantify the relative distance in groups of loops that included both length matched and mismatched pairs. Loops were aligned as described above to a high-resolution template, then raw DTW scores calculated between the coordinates using the &#x2018;dtaidistance&#x2019; library in python (<xref ref-type="bibr" rid="B41">41</xref>). Normalisation was applied by squaring the score, dividing by the number of atoms being compared, and taking the square root (this resulted in a DTW score that is equivalent to RMSD when lengths are matched). The squared score was divided by the maximum number of atoms (i.e., for length 9 versus 8, the score was divided by 27 to account for 9 residues, each comprised of two carbon and one nitrogen backbone atoms).</p>
</sec>
<sec id="s2_10">
<title>Density based clustering</title>
<p>The pairwise distance matrices of RMSD or DTW values contain information that represents the structural relationships between all loop conformations in an analysis. To identify clusters of loops with similar conformations within this high-dimensional data, we employed density-based clustering, using the density-based spatial clustering of applications with noise (DBSCAN) function from the scikit-learn library (SK learn) (<xref ref-type="bibr" rid="B42">42</xref>). The DBSCAN function takes the input distance matrix and two parameters: the minimum number of data points required to form a cluster (min points) and the minimum distance between any two points in the same cluster (epsilon). The number and size of clusters identified in each analysis is highly sensitive to these values and must be optimised based on the input data. Therefore, we systematically calculated these values in a consistent manner for all analyses, allowing us to find the maximum number of clusters within high-dimensional space and assess whether any cluster related to a novel canonical form. Min point values were calculated by taking the square root of either the number of loops being compared, or the same number divided by 2. For epsilon values we performed a K-nearest neighbours (KNN) analysis on the distance matrix for values of K ranging from 2 to 5. For each value of K, the elbow point was taken from a scree plot of all KNN distances, and this elbow point value used as epsilon in subsequent DBSCAN analyses. This resulted in four separate DBSCAN analyses (one for each value of K from 2 to 5). To select the most appropriate analysis for cluster inspection and visualisation, we matched the K value used to determine epsilon with the number of clusters identified by DBSCAN. If there was no match, then the next closest match was taken for the highest value of K. While dominant clusters were often evident and easily found using multiple values of epsilon, the impact of optimising these parameters was most apparent when identifying smaller or overlapping clusters. We call the clusters generated in this way DBS clusters (short for DBSCAN-based selection). Code is available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/oxpig/OAS-CanonicalForms">https://github.com/oxpig/OAS-CanonicalForms</ext-link>
</p>
</sec>
<sec id="s2_11">
<title>Inspection of density-based cluster structures and sequences</title>
<p>We manually inspected the loops comprising each DBS cluster by visualisation of both structures and sequences. This allowed us to assess the structural difference between each cluster and relate the sequence logos back to the defined sequences of PyIgClassify2 canonical clusters. The aligned 3D loops that were assigned to each DBS cluster were visualised using PyMol (<xref ref-type="bibr" rid="B43">43</xref>). We selected random samples of up to 20 loops from the predicted structures, all of which belonged to a specific DBS cluster. Samples were coloured according to cluster membership and viewed in the same frame. These loops were presented in multiple orientations to highlight backbone differences that led to distinct cluster assignment. For sequence logo plots, sequences from the predicted structures which had been assigned to a DBS cluster were plotted in R using the ggseqlogo package (<xref ref-type="bibr" rid="B44">44</xref>). Logo plots are shown in the bitwise format (opposed to the proportion format) to maximise identification of the dominant amino acid enrichments and motifs specific to each cluster.</p>
</sec>
<sec id="s2_12">
<title>Multidimensional scaling visualisation</title>
<p>To simplify the complex high-dimensional pairwise distance matrix and allow for easy visualisation, we applied multidimensional scaling (MDS) to create a 2D representation. The axes of these plots are labelled as MDS1 and MDS2 and represent unitless scales that capture the spatial differences between data points. We used the parallelised MDS function from the &#x2018;lmds&#x2019; R package, before processing and plotting the output data using tidyverse packages (<xref ref-type="bibr" rid="B45">45</xref>). These plots were then annotated according to the DBS cluster membership of each data point, or the canonical cluster assignment of only the experimentally derived data points.</p>
</sec>
<sec id="s2_13">
<title>ABodyBuilder2 out-of-domain experiments</title>
<p>Out-of-domain experiments involved the removal of all data points related to a specific CDR length, or a specific canonical cluster, from both the training and test datasets of ABB2. The model was then trained on this modified dataset from scratch. The criteria for removing data from ABB2 training samples involved dividing the MDS map into quadrants and selecting the quadrant with the most distinct canonical cluster, i.e. clearly separated from other data points. All experimental data points in this quadrant were excluded. To ensure the removal of all relevant data points from the training data, any samples defined as the excluded canonical form by PyIgClassify2 without an electron density cutoff (using the &#x2018;cluster_nocutoff&#x2019; flag) were also eliminated.</p>
</sec>
<sec id="s2_14">
<title>Data inclusion experiments</title>
<p>For out-of-domain experiments where small numbers of the excluded canonical form were reintroduced into the training data, we first added the highest-resolution datasets identified as representative of the missing canonical form (determined by the &#x2018;is_representative&#x2019; flag in PyIgClassify2). Subsequently, we progressively reintroduced the next highest-resolution datasets not designated as representative but still belonging to the high-confidence canonical cluster.</p>
</sec>
<sec id="s2_15">
<title>Retraining of ABodyBuilder2</title>
<p>The original ABodyBuilder2 model consists of an ensemble of four models each trained independently. To make a prediction the outputs from each model are averaged and the prediction closest to average is selected as the final output. Of the four models, one utilised a 128-dimensional embedding and three utilised a 256-dimensional embedding. Models were trained until no further improvement was seen in the validation loss after 100 epochs.</p>
<p>To facilitate the training of multiple models, for the initial experiments in this study we retrained a single model with a 128-dimensional embedding (not an ensemble). Each model was trained on either a Nvidia GeForce GTX-1080 Ti GPU or Quadro RTX 6000/8000 GPUs for 150-340 epochs for each training stage (see training methods <xref ref-type="bibr" rid="B20">20</xref>), continuing until no further improvement in validation loss was observed after 50 epochs (half the number used to train original ABB2 models). For specific retraining experiments (those used to confirm data inclusion thresholds important for prediction), an ensemble of models was created, each consisting of one model with a 128-dimensional embedding and three models with a 256-dimensional embedding. Each model followed the original ABB2 protocol (training until no improvement after 100 epochs). This process allowed us to carry out a larger number of experimental runs and only build full models when we had identified the data cutoffs that significantly affected prediction accuracy (assessed by RMSD between predictions and experimental data points).</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s3_1">
<title>Dominant CDR lengths in paired sequence space are matched by comparable distributions in structural data</title>
<p>We predicted the structures of ~1.5M paired antibody sequences from OAS (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B32">32</xref>) using ABodyBuilder2 (ABB2) (<xref ref-type="bibr" rid="B20">20</xref>), a state of the art deep learning antibody structure predictor. We examined this structural space for evidence of novel canonical forms.</p>
<p>We analysed the length distributions of the CDRL1, CDRL2, CDRL3, CDRH1 and CDRH2 loops in this dataset by both absolute frequency (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>) and by non-redundant CDR sequence frequency (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>). This revealed that many loops belonging to a specific length were dominated by a smaller number of unique sequences. For example, CDRL1 IMGT length 6 had a frequency of 753,690 in 1.49M, of which only 21,700 were unique. Therefore, we decided to focus our structural analysis on the CDR loops and length combinations (e.g. CDRL3 loops of length 9, after this point referred to as CDRL3-Len-9) which had the highest number of non-redundant sequences within the dataset (bars marked with an asterisk in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>, details and numbers given in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Dominant CDR lengths in paired sequence space are matched by comparable distributions in structural data. Frequency distributions of sequences present in OAS by CDR and loop length, for the total number, including all redundant sequences <bold>(A)</bold>, and unique sequences only <bold>(B)</bold>. Asterisks above certain bars in <bold>(B)</bold> indicate the lengths with the most unique sequences, the predicted structures of which were analysed. <bold>(C, D)</bold> show data for the antibodies used to develop ABB2. The loop length frequency for each CDR including all structural units (all copies in the asymmetric unit) which have information in PyIgClassify2 (<xref ref-type="bibr" rid="B28">28</xref>) are shown <bold>(C)</bold>. Breakdown of canonical form assignments for corresponding CDR loops present in structures of ABB2 training data <bold>(D)</bold>. Each colour within a bar represents a distinct canonical form, red portions indicate loops that could not be assigned to any canonical form with high confidence in PyIgClassify2 analyses (not labelled with a number in the colour legend).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1352703-g001.tif"/>
</fig>
<p>The length distributions of the experimentally derived SAbDab (<xref ref-type="bibr" rid="B38">38</xref>, <xref ref-type="bibr" rid="B39">39</xref>) structures used to develop ABB2 are shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1C</bold>
</xref>. Only structures from the train, test and validation datasets which had information on canonical forms detailed in PyIgClassify2 were analysed (<xref ref-type="bibr" rid="B28">28</xref>) (38 datasets used in development of ABB2 were not categorised in PyIgClassify2). The CDR loop and length combinations taken forward for further analysis were also enriched in the structural units used to train ABB2 (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1C</bold>
</xref>), with the minimum number of examples seen by ABB2 during training being 268 for CDRH1-Len-9.</p>
<p>We next analysed the high confidence PyIgClassify2 canonical form assignments of each CDR loop length marked for further investigation by plotting the proportions of each canonical form within all experimental units (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1D</bold>
</xref>, an experimental unit refers to the fact that one PDB file may have multiple copies in the asymmetric unit). This analysis demonstrated a similar bias, with a single canonical cluster dominating over 50% of assignments for 13 out of the 17 CDR loop and length combinations. Some canonical forms had a very small proportion of examples contained in the ABB2 development data, with the minimum number being 8 examples for CDRH1-Len-8 (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1D</bold>
</xref>). The biases in both the length and canonical clusters distributions of the experimental data indicate that some data-poor areas may benefit from augmentation with predicted structures.</p>
<p>Having selected the CDRs and lengths which dominated our predicted structures, we next built structural clusters for these and explored whether the predicted structures gave rise to new canonical forms or provided insights which were not evident from experimental data alone.</p>
</sec>
<sec id="s3_2">
<title>Predicted structures fall into dense regions of conformational space defined by existing canonical forms</title>
<p>We created a map of the structural space for each of the dominant CDR loop and length combinations, identified above, using the predicted antibody structural data. Each map was analysed to find clusters of CDR loops that shared the same backbone conformation. If a SAbDab structure belonged to a cluster this allowed us to annotate the clusters canonical form according to PyIgClassify2. These annotated maps of canonical form structural space enabled us to navigate the predicted structural space and identify highly occupied regions of space not currently defined by a canonical form.</p>
<p>A full description of how these structural space maps were generated is given in the methods. In brief, each map is a 2D representation of the 3D clustered space of a CDR type at a given length. Data points representing loops from both experimental and predicted structures are coloured by their DBS cluster membership. All data points which are not assigned to a DBS cluster are coloured black. For canonical form annotation the experimental data points are coloured according to their PyIgClassify2 high confidence canonical cluster assignment. Any loops that do not belong to a high confidence PyIgClassify2 cluster (defined by an asterisk in the canonical cluster label) are coloured in red.</p>
<p>
<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2A, B</bold>
</xref> show the structural space map for CDRL1-Len-6. The projections are overlaid with either DBS cluster membership information (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2A</bold>
</xref>), or canonical cluster classifications from PyIgClassify2 (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2B</bold>
</xref>). The sequences of the loops which comprised these clusters are visualised using logo plots (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2C</bold>
</xref>) to identify the motifs and amino acid enrichments which should match to the canonical sequence motifs described in PyIgClassify2 (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). Samples of loops were also inspected in 3D to assess differences in backbone conformations that give rise to the distinct clusters (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2D</bold>
</xref>).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Predicted Structures fall into dense regions of conformational space defined by existing canonical classes. Analysis of CDRL1 length 6 loops. Multi-dimensional scaling plots derived from pairwise RMSD data <bold>(A, B)</bold>. Data points in <bold>(A)</bold> are coloured according to density-based clustering (DBS) membership, with the cluster centroid marked by an <bold>X</bold>. Those data points which do not belong to a DBS cluster in <bold>(A)</bold> are coloured black. In <bold>(B)</bold> all experimental data points present in the MDS analysis are coloured according to their high confidence PyIgClassify2 canonical form annotation. Any loops that do not belong to a high confidence PyIgClassify2 cluster (defined by an asterisk in the label, e.g. L1-11-*) are coloured in red. The predicted data points are coloured in black and underly the experimental annotations. Logo plots are shown for all sequences in each DBS cluster <bold>(C)</bold>. For a sample of 20 loops in each cluster, the framework aligned backbone conformations are shown in three different orientations <bold>(D)</bold>. The logo plots <bold>(C)</bold> and backbone <bold>(D)</bold> colours of blue, green and red correspond to the numbered DBS cluster colours in <bold>(A)</bold>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1352703-g002.tif"/>
</fig>
<p>For many of the CDR and length combinations analysed in this way, the clusters arising from predicted structures aligned well with the dominant clusters of experimentally defined canonical forms and did not highlight any new areas of density without canonical cluster assignment (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;1</bold>
</xref>; <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). Inspection of loop alignments revealed that our RMSD and DBS analysis method could distinguish between backbone kinks, peptide flips and minor variations that equated to less than 1 &#xc5; RMSD between data points (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures&#xa0;2A&#x2013;C</bold>
</xref>). These minor differences in conformation were also detected by the dihedral angle metric used to compile PyIgClassify2 clusters resulting in similar global divisions of structural space. Before exploring areas that were not accounted for by existing definitions of canonical forms, we next inspected any inconsistencies between our RMSD/DBS analysis and PyIgClassify2.</p>
</sec>
<sec id="s3_3">
<title>Differences between PyIgClassify2 definitions and density based structural clusters</title>
<p>While most DBS clusters detected in our analysis could be mapped to experimental data points that adhered to a high confidence canonical form, several of the more subtle PyIgClassify2 definitions were assimilated into a single DBS cluster. We investigated these assimilated data points to assess whether our method was missing important conformational differences.</p>
<p>The loop which best exemplified this was CDRL3-Len-8 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;1A</bold>
</xref>). Our analysis pipeline identified two DBS clusters, the centroids of which were 1.45 &#xc5; apart and had distinct sequence motifs (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;1A</bold>
</xref>). Inspection of the PyIgClassify2 canonical clusters demonstrated that cluster 1 (coloured blue in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;1A</bold>
</xref>) defined by the logo motif of QQYysxxT was subdivided into two canonical clusters, of which one had a proline at position 7 (see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, canonical forms L3-8-1 and L3-8-3). We reran our DBS clustering method using an alternate min points term (square root of N data points, opposed to square root of N/2) and found this was able to subdivide the major cluster into two distinct clusters (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;2E</bold>
</xref>) distinguished by the proline at position 7 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;2F</bold>
</xref>). These two clusters are only 0.4 &#xc5; apart and exhibited a large degree of overlap in conformational space (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;2G</bold>
</xref>).</p>
<p>We found additional examples of this within the clusters of loops for CDRL1-Len-6 (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2B</bold>
</xref>) and CDRH2-Len-8 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;1C</bold>
</xref>) where the effect was apparent from the smaller number of DBS clusters annotated by a larger number of canonical cluster labels. However, these were often canonical forms assigned to a comparatively low proportion of experimental data points, with more dominant canonical clusters showing greater overlap with our analyses (see later figures). We reasoned that such subtle shifts in conformation would not serve as strong evidence of extrapolation, despite being valid definitions of canonical forms from the dihedral angle perspective. Therefore, their detection was not crucial to our exploratory search for new knowledge arising from structure prediction tools.</p>
</sec>
<sec id="s3_4">
<title>Predicted structures enrich the experimental landscape revealing subdivisions of existing classes with defined sequence motifs</title>
<p>Having confirmed that our clustering pipeline was able to pick out major differences in loop conformations across large datasets, we next investigated the structural clusters and sequence logos which did not sit within with PyIgClassify2 defined canonical forms. These ambiguous clusters could be divided into two categories, those which contained experimental data points defined by a canonical form but could be further subdivided into new clusters with distinct sequence motifs and loop conformations, and those which contained experimental data points that were not assigned to a canonical form.</p>
<p>Firstly, the enrichment of the existing structural space with predicted structures led to subdivisions of existing canonical clusters. For example, in CDRH1-Len-8 (<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3A, B</bold>
</xref>) and CDRH1-Len-10 (<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3C, D</bold>
</xref>) as well as CDRL1-Len-11 and CDRH2-Len-7 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures&#xa0;3A, B</bold>
</xref>), we observed DBS clusters with distinct amino acid motifs and loop conformations. These were resolved from the increased structural dataset. The four subclusters observed in CDRH1-Len-8 (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>) and derived from the PyIgClassify2 canonical form H1-13-5 (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>, for motifs and length comparisons see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>) showed different amino acid patterns at position 2, 4 and 5 of the loops. These subclusters had conformations which differed by RMSD of between 0.79-1.43 &#xc5; for cluster centroids. Meanwhile the two subclusters identified from the canonical form H1-15-2 (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3D</bold>
</xref>; <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>) in CDRH1-Len-10 loops exhibited a difference of 1.09 &#xc5; and had sequence patterns that differed at six of the ten positions (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3C</bold>
</xref>).</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Subdivisions of existing canonical classes into clusters with distinct sequence motifs and backbone conformations. For CDR loops of CDRH1 length 8 <bold>(A, B)</bold> and CDRH1 length 10 <bold>(C, D)</bold>, distinct DBS clusters contained multiple experimental data points sharing the same canonical form. <bold>(A, C)</bold> show colour coded DBS cluster logo plots, MDS plots coloured by DBS cluster membership and framework aligned backbone conformations of a sample of 20 loops from each cluster in three different orientations. Data points in the MDS plots which do not belong to a DBS cluster are coloured black. <bold>(B, D)</bold> show all experimental data points present in the MDS analysis coloured according to their high confidence PyIgClassify2 canonical form annotation. Any loops that do not belong to a high confidence PyIgClassify2 cluster (defined by an asterisk in the label, e.g., H1-13-* and H-15-*) are coloured in red. The predicted data points are coloured in black and underly the experimental annotations.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1352703-g003.tif"/>
</fig>
<p>For all four CDR loops where predictions gave rise to sub clusters, the RMSD between new cluster centroids were often close to the mean value of each analysis (range of mean values from all pairwise comparisons: 0.96 &#x2013; 1.03 &#xc5;, range of distances between cluster centroids: 0.27 &#x2013; 1.64 &#xc5;) and originated from examples present in the training data. Hence the novel canonical classes arising here did not come from generalisation or extrapolation, but simply by statistical power - the increased sensitivity of density-based clustering on the far larger structural set (number of experimental data units versus predicted structures for CDRL1-Len-11: 768 vs 6,992, CDRH1-Len-8: 5,196 vs 61,617, CDRH1-Len-10: 314 vs 2,500 and CDRH2-Len-7: 1298 vs 27,769).</p>
</sec>
<sec id="s3_5">
<title>Enrichment of unassigned areas of structural space defines new canonical forms within heterogeneous sequences</title>
<p>The second set of novel clusters identified related to dense areas of predicted structural space where a smaller number of experimental structures existed but were defined as &#x201c;unassigned&#x201d; to any canonical form in PyIgClassify2. For example, for CDRL1-Len-7 a cluster made up of 909 predicted data points (and 32 experimental data points) was identified which was distinct from the centroid of two existing canonical clusters by RMSD values of 3.94 and 4.31 &#xc5; respectively (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>). This &#x201c;new&#x201d; canonical form has a sequence motif with a strong preference for SGH at positions 1-3 of the loop, in contrast to QSV in both existing forms (see logo plot in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>, PyIgClassify2 annotations in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref> and corresponding motifs in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). A second example, CDRL1-Len-9 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4C</bold>
</xref>), had fewer experimental structures within that area (13 were present in training data) and comprised of 673 predictions. The central motif of INV at positions 3-5 showed no overlap with the enriched residues at the same positions within the three existing canonical forms (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4D</bold>
</xref>, see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>), with RMSD values between the corresponding cluster centroids of 2.99-3.51 &#xc5;. Both observations were enabled by increased population of structural space with ABB2 predictions from heterogeneous sequences, however they are not evidence of extrapolation given their origins in the training data.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Novel canonical classes revealed by structure prediction. Breakdown of analyses where regions of structural space populated by *unassigned experimental data (denoted by red data points with an asterisk in the PyIgClassify2 annotations) resolved to DBS clusters with distinct sequence logo plots for CDRL1 length 7 <bold>(A, B)</bold> and CDRL1 length 9 <bold>(C, D)</bold>. <bold>(A, C)</bold> show colour coded DBS cluster logo plots, MDS plots coloured by DBS cluster membership and framework aligned backbone conformations of a sample of 20 loops from each cluster in three different orientations. Data points in the MDS plots which do not belong to a DBS cluster are coloured black. <bold>(B, D)</bold> show all experimental data points present in the MDS analysis coloured according to their high confidence PyIgClassify2 canonical form annotation. Any loops that do not belong to a high confidence PyIgClassify2 cluster (defined by an asterisk in the label, e.g., L1-12-* and L-14-*) are coloured in red. The predicted data points are coloured in black and underly the experimental annotations.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1352703-g004.tif"/>
</fig>
</sec>
<sec id="s3_6">
<title>New forms exemplify length independent canonical classes arising from somatic hypermutation</title>
<p>Within the CDRL3 loops we found two further examples of highly populated DBS clusters that did not fit with any PyIgClassify2 definitions. These clusters were identified in the analyses of CDRL3 loops of lengths 10 and 11. Both areas of density contained experimental structures that were classified as unassigned to any high confidence canonical cluster by PyIgClassify2 (CDRL3-10 <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5A,B</bold>
</xref>, and CDRL3-11 <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5D, E</bold>
</xref>).</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>ABB2 can generalise across CDRL3 loops which differ in length by one amino acid. Novel DBS clusters were identified for CDRL3 length 10 <bold>(A&#x2013;C)</bold> and CDRL3 length 11 <bold>(D&#x2013;F)</bold>. <bold>(A, D)</bold> show experimental data points coloured according to their high confidence PyIgClassify2 canonical form annotation. Any loops that do not belong to a high confidence PyIgClassify2 cluster (defined by an asterisk in the label, e.g. L3-10-* and L3-11-*) are coloured in red. The predicted data points are coloured in black and underly the experimental annotations. Data points in <bold>(B, E)</bold> are coloured according to density-based clustering (DBS) membership, with the data points which do not belong to a DBS cluster coloured black. The areas circled in yellow on all four MDS plots <bold>(A, B, D, E)</bold> relate to the DBS clusters of predicted data points, containing experimental data not assigned to any canonical form that were likely to have arisen from length independence. Logo plots were shown with an arrow indicating the high entropy position with no consistent enrichment <bold>(C, F)</bold> and likely somatic insertion into a shorter canonical cluster (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;5</bold>
</xref>). Dynamic time warping analysis permitted quantification of cluster distances between CDR loops of different length as well as clusters of the same length. Clusters of the same length for CDRL3 length 10 are compared by visualising the backbone atoms and sequence logo plots <bold>(G)</bold>, while the novel cluster (coloured yellow) is compared against the proposed origin cluster that the short length of 9 in <bold>(H)</bold>. The DTW distance between cluster centroids is given below each logo plot. CDRL11 clusters of the same length are compared in <bold>(I)</bold>, then the novel cluster and proposed origin cluster in CDRL3 length 10 are compared in <bold>(J)</bold>. Out-of-domain experiments were carried out by retraining ABB2 in the absence of all experimental data points for each of the CDRL3 lengths 8, 9 and 10 [<bold>(K&#x2013;M)</bold>, also see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;6</bold>
</xref>]. Boxplots (median and upper and lower quartiles) and dot plots of RMSD values between predictions and ground truth for each starved model versus the original ABB2 ensemble are compared <bold>(K)</bold>, each dot corresponds to the RMSD value of one comparison. The global conformational space of predictions on withheld data points specific to each model are shown for the CDRL3 length 9 <bold>(L)</bold> and CDRL3 length 10 <bold>(M)</bold> starved models. The separation of data points according to canonical form classification was compared to true conformational space and ABB2 ensemble predictions in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;6</bold>
</xref> (for each length MDS calculation was performed all data points in the same analysis to allow comparison).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1352703-g005.tif"/>
</fig>
<p>However, these loop shapes are potentially derived from somatic hypermutation (SHM) insertions into CDR loop sequences classified as canonical forms at a shorter length (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures&#xa0;4A&#x2013;C</bold>
</xref>). These were evident from inspection of logo plots of CDRL3-Len-10 cluster 3 position 8 (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5C</bold>
</xref>), and CDRL3-Len-11 clusters 1 and 5 at position 9 (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5F</bold>
</xref>), where the motif was nearly identical to that of a highly populated cluster in the CDR one amino acid length below (PyIgClassify2 canonical forms of L3-9-2 and L-10-cis78-1, see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). The SHM insertions were clearly visible on each logo plot as they resulted in no consensus amino acid enrichment at a fixed position in the loop (positions are marked by an arrow in <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5C, F</bold>
</xref> respectively).</p>
<p>Given we could identify the corresponding canonical cluster at the shorter length (we termed this the &#x2018;origin cluster&#x2019;) (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;4</bold>
</xref>), we decided to quantify and compare the conformation differences between the two sets of loops. To obtain distance scores that could be used for comparison, both for loops of the same length and differing lengths, we substituted pairwise RMSD calculations with dynamic time warping (DTW) calculations which can be performed on coordinate arrays of differing dimensions (see methods for details).</p>
<p>We represented the structural relationships between CDRL3 loops of length 9 and 10 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;5A</bold>
</xref>), and CDRL3 loops of length 10 and 11 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;5B</bold>
</xref>) using DTW. The cluster of loops representing the novel conformation in CDRL3-10 (cluster 4: QQYxxxPxxT, coloured yellow in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;5A</bold>
</xref>) was closest in 3D space (DTW distance between cluster centroids of 0.68 &#xc5;) to a cluster composed of CDRL3-Len-9 loops (cluster 1: QQYysxxxT, coloured blue in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;5A</bold>
</xref>), and only 1.21 &#xc5; away from the proposed origin cluster (CDRL3-Len-9 cluster 2: QQyxxxPxT, coloured red in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;5A</bold>
</xref>). These distances were less than, or comparable to both clusters present at the same length of 10 (distances of 1.83 &#xc5; and 1.12 &#xc5; apart respectively).</p>
<p>The same effect was more pronounced for the novel conformation in CDRL3-Len-11 (cluster 5: QQYxxxPPxxT, coloured pink in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;5B</bold>
</xref>). Here the centroids of clusters found at the same length were 2.48 &#xc5; and 2.94 &#xc5; away, while the cluster at the shorter length was only 2.07 &#xc5; away. Visual inspection of the loop backbones from different clusters shows how similar conformations of mismatched lengths (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5H, J</bold>
</xref>) can be closer in 3D space than matched lengths from different DBS clusters (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5G, I</bold>
</xref>).</p>
<p>These observations fit with the previously described idea of length independence in canonical forms (<xref ref-type="bibr" rid="B26">26</xref>), where the closest partner of a structural cluster, or CDR loop, is another cluster, or loop, present at a different length. We hypothesise that the high frequency of predicted structures derived from heterogenous sequences in OAS altered by SHM, helped to reveal these length independent patterns.</p>
</sec>
<sec id="s3_7">
<title>ABB2 can generalise across CDRL3 loops which differ in length by one amino acid</title>
<p>To explicitly test whether the training and test data points with a specific CDR length influence the predictions of CDR loops of a different length we performed several out-of-domain experiments. These involved modifying the ABB2 training and test data to remove all data points containing CDRL3 loops length of 8, 9 or 10 (one length per experiment). In each case a new instance of the ABB2 model was trained on a reduced dataset. The resulting models were used to make predictions from sequences of the withheld length which could be assessed individually for accuracy, and together for occupancy of structural space.</p>
<p>The first model tested was trained in the absence of all 392 datapoints (referring to all copies in the asymmetric unit of each PDB file) that had CDRL3 length 8 loops. Prediction accuracy was poor (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures&#xa0;6A, B</bold>
</xref>), with median RMSD values (prediction versus ground truth) of 1.41 &#xc5; compared to 0.46 &#xc5; for the fully trained ABB2 model (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5K</bold>
</xref>). There was no clear separation of canonical clusters in conformational space, with median values for each form above 1 &#xc5; from the ground truth structure (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;6B</bold>
</xref>).</p>
<p>In contrast the model trained in the absence of CDRL3-Len-9 data had better prediction accuracy on the withheld structures (3865 datapoints, median RMSD 1.05 &#xc5;) but still worse than the fully trained ABB2 ensemble (median RMSD 0.46 &#xc5;). However, the MDS representation of conformational space showed early separation of data points defined by similar canonical clusters (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5L</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures&#xa0;6C, D</bold>
</xref>), indicating some rationalisation of the sequence to structure relationship via length offset data. Accuracy for the CDRL3-Len-10 model was the worst (median RMSD 1.66 &#xc5; <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5K</bold>
</xref>), however a higher standard deviation reflected the correct separation of global conformational space for data points of some canonical forms where loops were close to 1 &#xc5; RMSD from the ground truth structure (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5M</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures&#xa0;6E, F</bold>
</xref>).</p>
<p>There are only 9 data points of CDRL3 length 7, and this may explain why all predictions of CDRL3 length 8 for the starved model fell into a single cluster. In contrast, models starved of CDRL3 lengths 9 and 10 were able to separate some predictions into areas of conformational space close their ground truth structure. This may have been due the abundance of data points either side of the missing length in training data. These experiments, in addition to the structural overlap of predictions of different lengths, provided evidence of generalisation by ABB2 with origins in CDR length independence. Furthermore, these out-of-domain experiments serve as a powerful method to further explore the ability of deep learning-based structure prediction methods to extrapolate and find evidence of truly novel predictions.</p>
</sec>
<sec id="s3_8">
<title>Retraining whilst withholding canonical conformations highlights limited ability to extrapolate</title>
<p>Our analyses so far have not found any evidence of structural clusters representing novel conformations within the CDR loops of predicted antibody structures. Therefore, we set out to explicitly test whether ABB2 could predict a loop conformation not seen in the training data and without a parallel example at a different length. We ran out-of-domain experiments to train models in the absence of all examples of a specific canonical cluster and any close conformations (see methods). We focused on CDRL1 lengths 6-9 as these analyses showed the clearest cluster separation and the smallest proportion of data points that did not fall into a DBS cluster. This helped to avoid any ambiguity in the contents of the training data.</p>
<p>Separate models were trained for each withheld canonical class (numbers and training details given in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>). Each model was analysed as before, by predicting the structures of sequences in the withheld data and assessing the individual prediction accuracy as well as total occupancy of structural space.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Details of retraining in the absence of canonical clusters.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="left">Original<break/>Model</th>
<th valign="top" align="left">CDRL1-6<break/>Starved</th>
<th valign="top" align="left">CDRL1-7<break/>Starved</th>
<th valign="top" align="left">CDRL1-8<break/>Starved</th>
<th valign="top" align="left">CDRL1-9<break/>Starved</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">FV structural units seen in training (including all copies in the asymmetric unit)</td>
<td valign="top" align="left">5771</td>
<td valign="top" align="left">5459</td>
<td valign="top" align="left">5556</td>
<td valign="top" align="left">5671</td>
<td valign="top" align="left">5554</td>
</tr>
<tr>
<td valign="top" align="left">Dropped units (corresponding to a specific canonical form)</td>
<td valign="top" align="left">0</td>
<td valign="top" align="left">312</td>
<td valign="top" align="left">215</td>
<td valign="top" align="left">100</td>
<td valign="top" align="left">217</td>
</tr>
<tr>
<td valign="top" align="left">Total units with CDR and length tested (before withholding a specific form)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">2636</td>
<td valign="top" align="left">532</td>
<td valign="top" align="left">428</td>
<td valign="top" align="left">565</td>
</tr>
<tr>
<td valign="top" align="left">Remaining units with CDR tested (after removal)</td>
<td valign="top" align="left">_</td>
<td valign="top" align="left">2324</td>
<td valign="top" align="left">317</td>
<td valign="top" align="left">328</td>
<td valign="top" align="left">348</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Breakdown of the number of antibody variable fragment (FV) structural units used to train ABB2 and the subsequent &#x2018;starved&#x2019; models where all units containing a specific canonical form were withheld. The term structural unit is used to account for multiple copies being present in the asymmetric unit of the same PDB file. For each model, the total number of units seen in training is given, followed by the number of removed units. Then the total number of units related to the CDR being withheld is given (for example all units with CDRL1 IMGT length 6), followed by the number of those units remaining after withholding those related to a specific canonical form.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>For the &#x2018;starved&#x2019; model trained in the absence of CDRL1-Len-6 canonical cluster L1-11-3 (blue dots in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>, for sequence motif see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>), all predictions failed to match the ground truth conformation (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>) with mean (SD) RMSD difference of 1.54 (0.51) &#xc5;. The high standard deviation reflects how some predictions were closer (less than 0.5 &#xc5;) to the ground truth structure, however the majority adopted a similar conformation to the closest canonical form L1-11-4 (pink dots in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>, cluster centroid distance of 1.81 &#xc5;) rather than the more distant forms of L1-11-1 or L1-11-2 (green and mustard dots <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>, combined into one cluster by our method, centroid distance: 2.75 &#xc5;).</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Retraining whilst withholding canonical clusters highlights limited ability to extrapolate. Results of out-of-domain experiments where ABB2 was retrained in the absence of all experimental data points assigned to a specific canonical form in PyIgClassify2 (both high and low confidence) for CDRL1 lengths 6-9 <bold>(A&#x2013;D)</bold>. MDS plots show experimental data points coloured according to their high confidence PyIgClassify2 canonical form annotation. Any loops that do not belong to a high confidence PyIgClassify2 cluster (defined by an asterisk in the label, e.g., L1-11-*) are coloured in red. CDRL1 length 6 was retrained in the absence of all data points assigned to the PyIgClassify2 cluster &#x2018;L1-11-3&#x2019; [coloured blue in <bold>(A)</bold>]. For CDRL1 length 7 &#x2018;L1-12-2&#x2019; was dropped [coloured blue in <bold>(B)</bold>]. For CDRL1 length 8 cluster &#x2018;L1-13-3&#x2019; was dropped [coloured purple in <bold>(C)</bold>], and for CDRL1 length 9 cluster &#x2018;L1-14-1&#x2019; was dropped [coloured green in <bold>(D)</bold>]. For each panel, the MDS of experimental data points found in SAbDab are shown in the far-left panel. The MDS plot of ABB2 model predictions of the corresponding sequences are shown in the middle panel (labelled &#x2018;ABB2&#x2019;), and predictions of the starved model are shown in the right panel (labelled &#x2018;Starved&#x2019;). The area of structural space investigated through exclusion during training is highlighted in a red box. The boxplots (median and upper and lower quartiles) and dot plots on the far right of each panel indicate the RMSD values for each predicted data point from the starved model from its target structure in predictions from the fully trained ABB2 ensemble (top graph) or the ground truth structure (bottom graph) with each sub graph labelled accordingly.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1352703-g006.tif"/>
</fig>
<p>A more pronounced loss of the ability to extrapolate was observed for models starved of a canonical form in the remaining experiments using CDRL1-Len-7 (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>) and CDRL1-Len-9 (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6D</bold>
</xref>). Here all withheld data points fell into a more distant region of conformation space associated with mean prediction accuracies that were much lower, at mean (SD) RMSD values of 1.98 (0.09) &#xc5; RMSD for length 7, and 3.34 (0.19) &#xc5; for length 9.</p>
<p>For CDRL1-Len-7 the low prediction accuracy may have been due to very similar sequence motifs (DBS cluster: QSVSSSY, <italic>corresponding PyIgClassify2 cluster L1-12-1: RASqSVSSSYLa</italic>, versus DBS: QSVSSNY, <italic>PyIgClassify2 cluster L1-12-2: RASQSVSSNYLA, see</italic> <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>), as well as a small number of examples within the training data (63 examples of the withheld canonical form). In the case of CDRL1-Len-8, the model made predictions of L1-13-3 (purple dots <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6C</bold>
</xref>) that were in a similar region of structural space, however they were still 2.43 (0.19) &#xc5; away from ground truth values (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6C</bold>
</xref>).</p>
<p>These out-of-domain tests clearly demonstrated that the models trained in the absence of a canonical class were unable to recapitulate the correct conformations which could be predicted by the fully trained ABB2 ensemble. The extent of the distance between predictions and ground truth differed for each starved model and related to both sequence similarity as well as structural deviation. This led us to investigate whether prediction accuracy could be recovered by adding very small amounts of training examples back into the model.</p>
</sec>
<sec id="s3_9">
<title>Inclusion of a small number of examples in training is sufficient to recover predictive capacity</title>
<p>We set out to assess whether limited amounts of training data could recover missing predictions and thus quantify the level of representation needed to produce more accurate models. We chose CDRL1-Len-6 and CDRL1-Len-7 as these had the largest standard deviations in prediction accuracy (see box plots <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>). For each we progressively added increasing numbers of data points back into the initial out-of-domain tests resulting in separate models for each addition of data (CDRL1-Len-6 <xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7A, B</bold>
</xref> and CDRL1-Len-7 <xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7C, D</bold>
</xref>).</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Data points that make up less than 1% of training data are sufficient to recover predictive capacity. Out-of-domain experiments were performed as in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>, however for each model a specified number of experimental data points relating to the withheld canonical form were added into the training data (left to right for each panel, graphs are labelled with the percentages of data points added). Separate models were trained for each of the incrementally increasing data additions until the prediction accuracy came close to the fully trained ABB2 ensemble predictions and the ground truth experimental data. MDS plots for CDRL1 length 6 models <bold>(A)</bold> and CDRL1 length 7 <bold>(C)</bold>. The amount of data being included is labelled at the top of each panel as a percentage of the total number of unique structures out of all data points seen in training (see <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>). The corresponding ABB2 predictions and ground truth data are shown in the far-right MDS plots in each panel. RMSD distance values are plotted in <bold>(B, D)</bold> for the predicted data points relating to the dropped canonical form from each model relative to the ABB2 ensemble predictions (top graph) or ground truth data (bottom graph). For each model, the mean RMSD is given above the box plots, with standard deviation given in brackets. For <bold>(B, D)</bold>, the top, far-right graph RMSD values are zero as the ABB2 model predictions are being compared to itself, with the comparison of ABB2 to ground truth data shown below.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1352703-g007.tif"/>
</fig>
<p>As there were different total numbers of examples of CDRL1-Len-6 and CDRL1-Len-7 loops in training data (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>), the absolute number of PDB structures added back for each experiment did not represent the same proportion of datapoints. Therefore, we calculated the included loops as a percentage of all CDRL1 loops seen for each model (<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>). Using these values (<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>) and inspection of the corresponding model performance (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>), we could see that very small percentages of training data (less than 1%) were enough to allow the models to accurately recapitulate the cluster corresponding to the missing canonical form.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Inclusion data points as a proportion of total CDRL1 data units in training.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Number of included PDB structures of withheld canonical form</th>
<th valign="middle" align="left">CDRL1-6<break/>Number of structural units included (percentage of all data points of CDRL1)</th>
<th valign="middle" align="left">CDRL1-7<break/>Number of structural units included (percentage of all data points of CDRL1)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">All</td>
<td valign="middle" align="left">2636 (45.7%)</td>
<td valign="middle" align="left">532 (9.2%)</td>
</tr>
<tr>
<td valign="middle" align="left">0</td>
<td valign="middle" align="left">_</td>
<td valign="middle" align="left">_</td>
</tr>
<tr>
<td valign="middle" align="left">5</td>
<td valign="middle" align="left">9 (0.16%)</td>
<td valign="middle" align="left">8 (0.14%)</td>
</tr>
<tr>
<td valign="middle" align="left">10</td>
<td valign="middle" align="left">21 (0.38%)</td>
<td valign="middle" align="left">15 (0.27%)</td>
</tr>
<tr>
<td valign="middle" align="left">19</td>
<td valign="middle" align="left">_</td>
<td valign="middle" align="left">33 (0.59%)</td>
</tr>
<tr>
<td valign="middle" align="left">20</td>
<td valign="middle" align="left">43 (0.79%)</td>
<td valign="middle" align="left">_</td>
</tr>
<tr>
<td valign="middle" align="left">21</td>
<td valign="middle" align="left">_</td>
<td valign="middle" align="left">38 (0.68%)</td>
</tr>
<tr>
<td valign="middle" align="left">22</td>
<td valign="middle" align="left">_</td>
<td valign="middle" align="left">40 (0.72%)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>For inclusion experiments a specific number of PDB structures were gradually reintroduced to training data for a series of models. As PDB files may contain more than one structure in the asymmetric unit, the number of exact structural units is given for each experiment, as well as the corresponding percentage of all CDRL1 data points present in each training run.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>To check that these small proportions of data were enough to facilitate accurate predictions within the original ensemble architecture of ABB2, instead of a single model, we trained ensembles on two inclusion proportions (one above and one below the proportion where we first saw improvement ~ 0.6%, see <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>) and analysed the resulting final output (the average structure from all four predictions). This demonstrated that the ensemble models still failed to recapitulate at the lower percentage, while the higher percentages were sufficient for the ensemble to progressively populate the missing structural space despite still being below 1% of total examples of that CDR loop (RMSD plots marked as &#x2018;ensemble&#x2019; in <xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7B, D</bold>
</xref>).</p>
<p>These analyses underline the importance of sufficient data representation and suggest that even small numbers of datapoints can influence the predictive capacity drawn from large datasets. Ultimately the inability of models to truly extrapolate from physical principles means researchers must pay close attention to the contents of their datasets and continue to collect experimental data that explores lesser studied areas of structural space.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>In this study we analysed the predicted structures for paired sequences present in OAS generated by ABodyBuilder2 (ABB2) (<xref ref-type="bibr" rid="B20">20</xref>). Our data driven approach allowed identification of structural clusters within CDR loops of the same length and subsequent linkage to existing definitions of canonical forms. Our analyses aimed to explore the ability of structure predictors to enrich the experimentally defined landscape of canonical forms and identify novel conformations that would reflect generalisation or extrapolation arising from a deep learning method.</p>
<p>The augmentation of existing data with predicted structures enabled us to define new canonical clusters composed of heterogeneous CDR sequences which were united by the same loop backbone conformation and a sequence motif. These arose from both subdivision of areas of conformational space with uniform PyIgClassify2 annotation, as well as within highly populated areas of conformational space not assigned to any existing canonical form definition. Novel clusters of predicted loop conformations were also produced via the phenomenon of length independence (<xref ref-type="bibr" rid="B26">26</xref>). We observed areas of new density which had sequence enrichments identical to those of canonical forms at a shorter CDR length but contained a positionally fixed high entropy residue likely indicative of somatic hypermutation insertion. We analysed the distances between loop conformations at both the shorter and longer lengths by dynamic time warping. This revealed that different length loop clusters were indeed closer in conformational space than any of those of the same length. Out-of-domain experiments confirmed that ABB2 was able to generalise across loops of different lengths and suggested predictions were influenced by high frequency experimental datapoints seen in shorter CDR loops.</p>
<p>However, our analyses could not find new clusters which had no origin in training data and thus represented true extrapolation. Therefore, we performed further out-of-domain experiments by retraining our ABB2 structure predictor whilst withholding data points which belonged to specific canonical forms. These &#x2018;starved&#x2019; models were then challenged with correctly predicting the unseen CDR loop conformation. We found that ABB2 retrained in this way was unable to predict conformations not seen during training, but this inability could be resolved by inclusion of a small number of examples representing between 0.5-1.0% of the total data used in development. This suggests that effective prediction accuracy by structure predictors can be achieved for conformations even when they have very poor representation in the dataset.</p>
<p>Our study highlights important limitations regarding the current capabilities of deep learning structure prediction tools specific to the domain of immune receptor CDR loops. Whilst numerous studies have performed out-of-domain experiments and exploratory analysis on protein folds, the conformational space of CDR loops may offer greater challenges, particularly in regions that are inherently flexible or adopt distinct structures in bound and unbound states.</p>
<p>These challenges emphasise that if we wish to predict outside current known structure space, new structure prediction tools that have learnt the underlying rules which govern tertiary structure, instead of just patterns in the training data, will be required. While AlphaFold2 was heralded as a huge advance in structural biology and machine learning, the higher goal of building models that can capture biophysical laws has still not been reached (<xref ref-type="bibr" rid="B46">46</xref>, <xref ref-type="bibr" rid="B47">47</xref>). In the absence of architectures that can extrapolate, greater amounts of training data, particularly in regions of structure space with poor coverage, may help improve predictive accuracy. Our demonstration that a small number of examples can address gaps means that a critical mass of data could be achieved to overcome the limitations of current models. However, this does place a large burden on experimental researchers to collect more data.</p>
<p>Finally, our results focused on an area of structural immunology relatively abundant with data and analyses, that of antibodies. As T cell receptors become more important in both immunotherapy research and the clinic, a need to better classify and understand this protein for the purpose of structure prediction may supersede that of antibodies. Therefore, the questions posed in this study may take on more relevance in a field with a relative paucity of structure and paired sequencing data (<xref ref-type="bibr" rid="B48">48</xref>), as well as several unanswered questions on TCR loop flexibility and comparative conformational freedom (<xref ref-type="bibr" rid="B29">29</xref>).</p>
<p>The original purpose of canonical forms was their ability to predict structure from sequence, however these use cases have been superseded by the improved performance of ML methods for structure prediction. For experimental techniques such as X-ray crystallography to be rendered redundant in immunology we must have confidence that structure prediction algorithms faithfully replicate the most important region of immune receptors, the CDR loops.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are publicly available. This data can be found here: <uri xlink:href="https://doi.org/10.5281/zenodo.10280181">https://doi.org/10.5281/zenodo.10280181</uri>.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>AG-W: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. BA: Data curation, Investigation, Methodology, Software, Writing &#x2013; review &amp; editing. CD: Conceptualization, Funding acquisition, Investigation, Project administration, Resources, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by the Engineering and Physical Sciences Research Council (grant number EP/S024093/1) AG-W is funded by Exscientia and BA was funded by Roche. The funders were not involved in the study design, collection, analysis, interpretation of data, the writing of this article, or the decision to submit it for publication.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>The authors wish to thank Joao Diniz Brandao Gervasio and Carlos Outeiral Rubiera from the Oxford Protein Immunoinformatics Group for their helpful comments and discussions.</p>
</ack>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fimmu.2024.1352703/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fimmu.2024.1352703/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.pdf" id="SM1" mimetype="application/pdf"/>
</sec>
<fn-group>
<title>Abbreviations</title>
<fn fn-type="abbr">
<p>ABB2, ABodyBuilder2; AF2 AlphaFold2; AFM AlphaFold-Multimer; CDR, complementarity determining region; DBSCAN, density-based spatial clustering of applications with noise, DBS, DBSCAN-based selection; DTW, dynamic time warping; KNN, K-nearest neighbours; IMGT, the international ImMunoGeneTics information system; MDS, multidimensional scaling; OAS, observed antibody space; OPIG, Oxford protein informatics group; PDB, protein data bank; RMSD, root mean-squared deviation; SAbDab, the structural antibody database; SCFV, single chain variable fragment.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jumper</surname> <given-names>J</given-names>
</name>
<name>
<surname>Evans</surname> <given-names>R</given-names>
</name>
<name>
<surname>Pritzel</surname> <given-names>A</given-names>
</name>
<name>
<surname>Green</surname> <given-names>T</given-names>
</name>
<name>
<surname>Figurnov</surname> <given-names>M</given-names>
</name>
<name>
<surname>Ronneberger</surname> <given-names>O</given-names>
</name>
<etal/>
</person-group>. <article-title>Highly accurate protein structure prediction with AlphaFold</article-title>. <source>Nature</source>. (<year>2021</year>) <volume>596</volume>:<page-range>583&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baek</surname> <given-names>M</given-names>
</name>
<name>
<surname>DiMaio</surname> <given-names>F</given-names>
</name>
<name>
<surname>Anishchenko</surname> <given-names>I</given-names>
</name>
<name>
<surname>Dauparas</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ovchinnikov</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>GR</given-names>
</name>
<etal/>
</person-group>. <article-title>Accurate prediction of protein structures and interactions using a three-track neural network</article-title>. <source>Science</source>. (<year>2021</year>) <volume>373</volume>:<page-range>871&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.abj8754</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Akin</surname> <given-names>H</given-names>
</name>
<name>
<surname>Rao</surname> <given-names>R</given-names>
</name>
<name>
<surname>Hie</surname> <given-names>B</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>W</given-names>
</name>
<etal/>
</person-group>. <article-title>Evolutionary-scale prediction of atomic-level protein structure with a language model</article-title>. <source>Science</source>. (<year>2023</year>) <volume>379</volume>:<page-range>1123&#x2013;30</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.ade2574</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berman</surname> <given-names>HM</given-names>
</name>
<name>
<surname>Westbrook</surname> <given-names>J</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Gilliland</surname> <given-names>G</given-names>
</name>
<name>
<surname>Bhat</surname> <given-names>TN</given-names>
</name>
<name>
<surname>Weissig</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>The protein data bank</article-title>. <source>Nucleic Acids Res</source>. (<year>2000</year>) <volume>28</volume>:<page-range>235&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/28.1.235</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bordin</surname> <given-names>N</given-names>
</name>
<name>
<surname>Sillitoe</surname> <given-names>I</given-names>
</name>
<name>
<surname>Nallapareddy</surname> <given-names>V</given-names>
</name>
<name>
<surname>Rauer</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lam</surname> <given-names>SD</given-names>
</name>
<name>
<surname>Waman</surname> <given-names>VP</given-names>
</name>
<etal/>
</person-group>. <article-title>AlphaFold2 reveals commonalities and novelties in protein structure space for 21 model organisms</article-title>. <source>Commun Biol</source>. (<year>2023</year>) <volume>6</volume>:<fpage>1</fpage>&#x2013;<lpage>12</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s42003-023-04488-9</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>van Kempen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>SS</given-names>
</name>
<name>
<surname>Tumescheit</surname> <given-names>C</given-names>
</name>
<name>
<surname>Mirdita</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gilchrist</surname> <given-names>CLM</given-names>
</name>
<etal/>
</person-group>. <article-title>Fast and accurate protein structure search with Foldseek</article-title>. <source>Nat Biotechnol.</source> (<year>2023</year>) <fpage>1</fpage>&#x2013;<lpage>4</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41587-023-01773-0</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahdritz</surname> <given-names>G</given-names>
</name>
<name>
<surname>Bouatta</surname> <given-names>N</given-names>
</name>
<name>
<surname>Kadyan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Gerecke</surname> <given-names>W</given-names>
</name>
<name>
<surname>O&#x2019;Donnell</surname> <given-names>TJ</given-names>
</name>
<etal/>
</person-group>. <article-title>OpenFold: Retraining AlphaFold2 yields new insights into its learning mechanisms and capacity for generalization</article-title>. <source>bioRxiv</source> [Preprint] (<year>2022</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2022.11.20.517210</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Balestriero</surname> <given-names>R</given-names>
</name>
<name>
<surname>Pesenti</surname> <given-names>J</given-names>
</name>
<name>
<surname>LeCun</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Learning in high dimension always amounts to extrapolation</article-title>. <source>arXiv</source> [Preprint] (<year>2021</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2110.09485</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fannjiang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Listgarten</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Is novelty predictable</article-title>? <source>arXiv</source> [Preprint] (<year>2023</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2306.00872</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ruff</surname> <given-names>KM</given-names>
</name>
<name>
<surname>Pappu</surname> <given-names>RV</given-names>
</name>
</person-group>. <article-title>AlphaFold and implications for intrinsically disordered proteins</article-title>. <source>J Mol Biol</source>. (<year>2021</year>) <volume>433</volume>:<elocation-id>167208</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jmb.2021.167208</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tunyasuvunakool</surname> <given-names>K</given-names>
</name>
<name>
<surname>Adler</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Green</surname> <given-names>T</given-names>
</name>
<name>
<surname>Zielinski</surname> <given-names>M</given-names>
</name>
<name>
<surname>&#x17d;&#xed;dek</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Highly accurate protein structure prediction for the human proteome</article-title>. <source>Nature</source>. (<year>2021</year>) <volume>596</volume>:<page-range>590&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41586-021-03828-1</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chakravarty</surname> <given-names>D</given-names>
</name>
<name>
<surname>Porter</surname> <given-names>LL</given-names>
</name>
</person-group>. <article-title>AlphaFold2 fails to predict protein fold switching</article-title>. <source>Protein Sci</source>. (<year>2022</year>) <volume>31</volume>:<elocation-id>e4353</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/pro.4353</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schatz</surname> <given-names>DG</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Recombination centres and the orchestration of V(D)J recombination</article-title>. <source>Nat Rev Immunol</source>. (<year>2011</year>) <volume>11</volume>:<page-range>251&#x2013;63</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nri2941</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brack</surname> <given-names>C</given-names>
</name>
<name>
<surname>Hirama</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lenhard-Schuller</surname> <given-names>R</given-names>
</name>
<name>
<surname>Tonegawa</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>A complete immunoglobulin gene is created by somatic recombination</article-title>. <source>Cell</source>. (<year>1978</year>) <volume>15</volume>:<fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/0092-8674(78)90078-8</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alt</surname> <given-names>FW</given-names>
</name>
<name>
<surname>Yancopoulos</surname> <given-names>GD</given-names>
</name>
<name>
<surname>Blackwell</surname> <given-names>TK</given-names>
</name>
<name>
<surname>Wood</surname> <given-names>C</given-names>
</name>
<name>
<surname>Thomas</surname> <given-names>E</given-names>
</name>
<name>
<surname>Boss</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Ordered rearrangement of immunoglobulin heavy chain variable region segments</article-title>. <source>EMBO J</source>. (<year>1984</year>) <volume>3</volume>:<page-range>1209&#x2013;19</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/j.1460-2075.1984.tb01955.x</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Griffiths</surname> <given-names>GM</given-names>
</name>
<name>
<surname>Berek</surname> <given-names>C</given-names>
</name>
<name>
<surname>Kaartinen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Milstein</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Somatic mutation and the maturation of immune response to 2-phenyl oxazolone</article-title>. <source>Nature</source>. (<year>1984</year>) <volume>312</volume>:<page-range>271&#x2013;5</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/312271a0</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laserson</surname> <given-names>U</given-names>
</name>
<name>
<surname>Vigneault</surname> <given-names>F</given-names>
</name>
<name>
<surname>Gadala-Maria</surname> <given-names>D</given-names>
</name>
<name>
<surname>Yaari</surname> <given-names>G</given-names>
</name>
<name>
<surname>Uduman</surname> <given-names>M</given-names>
</name>
<name>
<surname>Vander Heiden</surname> <given-names>JA</given-names>
</name>
<etal/>
</person-group>. <article-title>High-resolution antibody dynamics of vaccine-induced immune responses</article-title>. <source>Proc Natl Acad Sci</source>. (<year>2014</year>) <volume>111</volume>:<page-range>4928&#x2013;33</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1323862111</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Regep</surname> <given-names>C</given-names>
</name>
<name>
<surname>Georges</surname> <given-names>G</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J</given-names>
</name>
<name>
<surname>Popovic</surname> <given-names>B</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>The H3 loop of antibodies shows unique structural characteristics</article-title>. <source>Proteins Struct Funct Bioinforma</source>. (<year>2017</year>) <volume>85</volume>:<page-range>1311&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/prot.25291</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Evans</surname> <given-names>R</given-names>
</name>
<name>
<surname>O&#x2019;Neill</surname> <given-names>M</given-names>
</name>
<name>
<surname>Pritzel</surname> <given-names>A</given-names>
</name>
<name>
<surname>Antropova</surname> <given-names>N</given-names>
</name>
<name>
<surname>Senior</surname> <given-names>A</given-names>
</name>
<name>
<surname>Green</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>Protein complex prediction with AlphaFold-Multimer</article-title>. <source>bioRxiv</source> [Preprint] (<year>2022</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2021.10.04.463034</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abanades</surname> <given-names>B</given-names>
</name>
<name>
<surname>Wong</surname> <given-names>WK</given-names>
</name>
<name>
<surname>Boyles</surname> <given-names>F</given-names>
</name>
<name>
<surname>Georges</surname> <given-names>G</given-names>
</name>
<name>
<surname>Bujotzek</surname> <given-names>A</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>ImmuneBuilder: Deep-Learning models for predicting the structures of immune proteins</article-title>. <source>Commun Biol</source>. (<year>2023</year>) <volume>6</volume>:<fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s42003-023-04927-7</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guloglu</surname> <given-names>B</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>Specific attributes of the VL domain influence both the structure and structural variability of CDR-H3 through steric effects</article-title>. <source>Front Immunol</source>. (<year>2023</year>) <volume>14</volume>. doi: <pub-id pub-id-type="doi">10.3389/fimmu.2023.1223802</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chothia</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lesk</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Tramontano</surname> <given-names>A</given-names>
</name>
<name>
<surname>Levitt</surname> <given-names>M</given-names>
</name>
<name>
<surname>Smith-Gill</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Air</surname> <given-names>G</given-names>
</name>
<etal/>
</person-group>. <article-title>Conformations of immunoglobulin hypervariable regions</article-title>. <source>Nature</source>. (<year>1989</year>) <volume>342</volume>:<page-range>877&#x2013;83</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/342877a0</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chothia</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lesk</surname> <given-names>AM</given-names>
</name>
</person-group>. <article-title>Canonical structures for the hypervariable regions of immunoglobulins</article-title>. <source>J Mol Biol</source>. (<year>1987</year>) <volume>196</volume>:<page-range>901&#x2013;17</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/0022-2836(87)90412-8</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>North</surname> <given-names>B</given-names>
</name>
<name>
<surname>Lehmann</surname> <given-names>A</given-names>
</name>
<name>
<surname>Dunbrack</surname> <given-names>RL</given-names>
</name>
</person-group>. <article-title>A new clustering of antibody CDR loop conformations</article-title>. <source>J Mol Biol</source>. (<year>2011</year>) <volume>406</volume>:<page-range>228&#x2013;56</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jmb.2010.10.030</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adolf-Bryfogle</surname> <given-names>J</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Q</given-names>
</name>
<name>
<surname>North</surname> <given-names>B</given-names>
</name>
<name>
<surname>Lehmann</surname> <given-names>A</given-names>
</name>
<name>
<surname>Dunbrack</surname> <given-names>RL</given-names>
</name>
</person-group>. <article-title>PyIgClassify: a database of antibody CDR structural classifications</article-title>. <source>Nucleic Acids Res</source>. (<year>2015</year>) <volume>43</volume>:<page-range>D432&#x2013;438</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gku1106</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nowak</surname> <given-names>J</given-names>
</name>
<name>
<surname>Baker</surname> <given-names>T</given-names>
</name>
<name>
<surname>Georges</surname> <given-names>G</given-names>
</name>
<name>
<surname>Kelm</surname> <given-names>S</given-names>
</name>
<name>
<surname>Klostermann</surname> <given-names>S</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Length-independent structural similarities enrich the antibody CDR canonical class model</article-title>. <source>mAbs</source>. (<year>2016</year>) <volume>8</volume>:<page-range>751&#x2013;60</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/19420862.2016.1158370</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wong</surname> <given-names>WK</given-names>
</name>
<name>
<surname>Georges</surname> <given-names>G</given-names>
</name>
<name>
<surname>Ros</surname> <given-names>F</given-names>
</name>
<name>
<surname>Kelm</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lewis</surname> <given-names>AP</given-names>
</name>
<name>
<surname>Taddese</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>SCALOP: sequence-based antibody canonical loop structure annotation</article-title>. <source>Bioinforma Oxf Engl</source>. (<year>2019</year>) <volume>35</volume>:<page-range>1774&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/bty877</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kelow</surname> <given-names>S</given-names>
</name>
<name>
<surname>Faezov</surname> <given-names>B</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Parker</surname> <given-names>M</given-names>
</name>
<name>
<surname>Adolf-Bryfogle</surname> <given-names>J</given-names>
</name>
<name>
<surname>Dunbrack</surname> <given-names>RL</given-names>
</name>
</person-group>. <article-title>A penultimate classification of canonical antibody CDR conformations</article-title>. <source>bioRxiv</source> [Preprint] (<year>2022</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2022.10.12.511988</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wong</surname> <given-names>WK</given-names>
</name>
<name>
<surname>Leem</surname> <given-names>J</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>Comparative analysis of the CDR loops of antigen receptors</article-title>. <source>Front Immunol</source>. (<year>2019</year>) <volume>10</volume>. doi: <pub-id pub-id-type="doi">10.3389/fimmu.2019.02454</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sivasubramanian</surname> <given-names>A</given-names>
</name>
<name>
<surname>Sircar</surname> <given-names>A</given-names>
</name>
<name>
<surname>Chaudhury</surname> <given-names>S</given-names>
</name>
<name>
<surname>Gray</surname> <given-names>JJ</given-names>
</name>
</person-group>. <article-title>Toward high-resolution homology modeling of antibody Fv regions and application to antibody-antigen docking</article-title>. <source>Proteins</source>. (<year>2009</year>) <volume>74</volume>:<fpage>497</fpage>&#x2013;<lpage>514</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/prot.22309</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kovaltsuk</surname> <given-names>A</given-names>
</name>
<name>
<surname>Leem</surname> <given-names>J</given-names>
</name>
<name>
<surname>Kelm</surname> <given-names>S</given-names>
</name>
<name>
<surname>Snowden</surname> <given-names>J</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
<name>
<surname>Krawczyk</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>Observed antibody space: A resource for data mining next-generation sequencing of antibody repertoires</article-title>. <source>J Immunol</source>. (<year>2018</year>) <volume>201</volume>:<page-range>2502&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.1800708</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Olsen</surname> <given-names>TH</given-names>
</name>
<name>
<surname>Boyles</surname> <given-names>F</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>Observed Antibody Space: A diverse database of cleaned, annotated, and translated unpaired and paired antibody sequences</article-title>. <source>Protein Sci Publ Protein Soc</source>. (<year>2022</year>) <volume>31</volume>:<page-range>141&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/pro.4205</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maier</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Martinez</surname> <given-names>C</given-names>
</name>
<name>
<surname>Kasavajhala</surname> <given-names>K</given-names>
</name>
<name>
<surname>Wickstrom</surname> <given-names>L</given-names>
</name>
<name>
<surname>Hauser</surname> <given-names>KE</given-names>
</name>
<name>
<surname>Simmerling</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>ff14SB: improving the accuracy of protein side chain and backbone parameters from ff99SB</article-title>. <source>J Chem Theory Comput</source>. (<year>2015</year>) <volume>11</volume>:<page-range>3696&#x2013;713</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acs.jctc.5b00255</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eastman</surname> <given-names>P</given-names>
</name>
<name>
<surname>Swails</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chodera</surname> <given-names>JD</given-names>
</name>
<name>
<surname>McGibbon</surname> <given-names>RT</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Beauchamp</surname> <given-names>KA</given-names>
</name>
<etal/>
</person-group>. <article-title>OpenMM 7: Rapid development of high performance algorithms for molecular dynamics</article-title>. <source>PloS Comput Biol</source>. (<year>2017</year>) <volume>13</volume>:<elocation-id>e1005659</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pcbi.1005659</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lefranc</surname> <given-names>M-P</given-names>
</name>
<name>
<surname>Lefranc</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>Antibody sequence and structure analyses using IMGT&#xae;: 30 years of immunoinformatics</article-title>. <source>Methods Mol Biol Clifton NJ</source>. (<year>2023</year>) <volume>2552</volume>:<fpage>3</fpage>&#x2013;<lpage>59</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-1-0716-2609-2_1</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dunbar</surname> <given-names>J</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>ANARCI: antigen receptor numbering and receptor classification</article-title>. <source>Bioinforma Oxf Engl</source>. (<year>2016</year>) <volume>32</volume>:<fpage>298</fpage>&#x2013;<lpage>300</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btv552</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>N</given-names>
</name>
<name>
<surname>Madden</surname> <given-names>TL</given-names>
</name>
<name>
<surname>Ostell</surname> <given-names>JM</given-names>
</name>
</person-group>. <article-title>IgBLAST: an immunoglobulin variable domain sequence analysis tool</article-title>. <source>Nucleic Acids Res</source>. (<year>2013</year>) <volume>41</volume>:<page-range>W34&#x2013;40</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkt382</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dunbar</surname> <given-names>J</given-names>
</name>
<name>
<surname>Krawczyk</surname> <given-names>K</given-names>
</name>
<name>
<surname>Leem</surname> <given-names>J</given-names>
</name>
<name>
<surname>Baker</surname> <given-names>T</given-names>
</name>
<name>
<surname>Fuchs</surname> <given-names>A</given-names>
</name>
<name>
<surname>Georges</surname> <given-names>G</given-names>
</name>
<etal/>
</person-group>. <article-title>SAbDab: the structural antibody database</article-title>. <source>Nucleic Acids Res</source>. (<year>2014</year>) <volume>42</volume>:<page-range>D1140&#x2013;1146</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkt1043</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname> <given-names>C</given-names>
</name>
<name>
<surname>Raybould</surname> <given-names>MIJ</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>SAbDab in the age of biotherapeutics: updates including SAbDab-nano, the nanobody structure tracker</article-title>. <source>Nucleic Acids Res</source>. (<year>2022</year>) <volume>50</volume>:<page-range>D1368&#x2013;72</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkab1050</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Honegger</surname> <given-names>A</given-names>
</name>
<name>
<surname>Pl&#xfc;ckthun</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Yet another numbering scheme for immunoglobulin variable domains: an automatic modeling and analysis tool</article-title>. <source>J Mol Biol</source>. (<year>2001</year>) <volume>309</volume>:<page-range>657&#x2013;70</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1006/jmbi.2001.4662</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meert</surname> <given-names>W</given-names>
</name>
<name>
<surname>Hendrickx</surname> <given-names>K</given-names>
</name>
<name>
<surname>Van Craenendonck</surname> <given-names>T</given-names>
</name>
<name>
<surname>Robberechts</surname> <given-names>P</given-names>
</name>
<name>
<surname>Blockeel</surname> <given-names>H</given-names>
</name>
<name>
<surname>Davis</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>DTAIDistance</article-title>. (<year>2020</year>). Available at: <uri xlink:href="https://zenodo.org/records/7158824">https://zenodo.org/records/7158824</uri>.</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedregosa</surname> <given-names>F</given-names>
</name>
<name>
<surname>Varoquaux</surname> <given-names>G</given-names>
</name>
<name>
<surname>Gramfort</surname> <given-names>A</given-names>
</name>
<name>
<surname>Michel</surname> <given-names>V</given-names>
</name>
<name>
<surname>Thirion</surname> <given-names>B</given-names>
</name>
<name>
<surname>Grisel</surname> <given-names>O</given-names>
</name>
<etal/>
</person-group>. <article-title>Scikit-learn: machine learning in python</article-title>. <source>J Mach Learn Res</source> (<year>2011</year>). Available at: <uri xlink:href="https://www.jmlr.org/papers/volume12/pedregosa11a/pedregosa11a.pdf">https://www.jmlr.org/papers/volume12/pedregosa11a/pedregosa11a.pdf</uri>.</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Delano</surname> <given-names>W</given-names>
</name>
</person-group>. <article-title>The pyMOL molecular graphics system</article-title>. (<year>2002</year>). Available at: <uri xlink:href="https://legacy.ccp4.ac.uk/newsletters/newsletter40/11_pymol.pdf">https://legacy.ccp4.ac.uk/newsletters/newsletter40/11_pymol.pdf</uri>.</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wagih</surname> <given-names>O</given-names>
</name>
</person-group>. <article-title>ggseqlogo: a versatile R package for drawing sequence logos</article-title>. <source>Bioinformatics</source>. (<year>2017</year>) <volume>33</volume>:<page-range>3645&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btx469</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wickham</surname> <given-names>H</given-names>
</name>
<name>
<surname>Averick</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bryan</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>W</given-names>
</name>
<name>
<surname>McGowan</surname> <given-names>L</given-names>
</name>
<name>
<surname>Fran&#xe7;ois</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Welcome to the tidyverse</article-title>. <source>J Open Source Softw</source>. (<year>2019</year>) <volume>4</volume>:<elocation-id>1686</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.21105/joss.01686</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Outeiral</surname> <given-names>C</given-names>
</name>
<name>
<surname>Nissley</surname> <given-names>DA</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>Current structure predictors are not learning the physics of protein folding</article-title>. <source>Bioinformatics</source>. (<year>2022</year>) <volume>38</volume>:<page-range>1881&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btab881</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buttenschoen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Morris</surname> <given-names>GM</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>PoseBusters: AI-based docking methods fail to generate physically valid poses or generalise to novel sequences</article-title>. <source>Chem Sci.</source> (<year>2024</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.1039/D3SC04185A</pub-id>
</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leem</surname> <given-names>J</given-names>
</name>
<name>
<surname>de Oliveira</surname> <given-names>SHP</given-names>
</name>
<name>
<surname>Krawczyk</surname> <given-names>K</given-names>
</name>
<name>
<surname>Deane</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>STCRDab: the structural T-cell receptor database</article-title>. <source>Nucleic Acids Res</source>. (<year>2018</year>) <volume>46</volume>:<page-range>D406&#x2013;12</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkx971</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>