<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mol. Biosci.</journal-id>
<journal-title>Frontiers in Molecular Biosciences</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mol. Biosci.</abbrev-journal-title>
<issn pub-type="epub">2296-889X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1204157</article-id>
<article-id pub-id-type="doi">10.3389/fmolb.2023.1204157</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Molecular Biosciences</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Understanding structure-guided variant effect predictions using 3D convolutional neural networks</article-title>
<alt-title alt-title-type="left-running-head">Ramakrishnan et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fmolb.2023.1204157">10.3389/fmolb.2023.1204157</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ramakrishnan</surname>
<given-names>Gayatri</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2273598/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Baakman</surname>
<given-names>Coos</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2358489/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Heijl</surname>
<given-names>Stephan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Vroling</surname>
<given-names>Bas</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>van Horck</surname>
<given-names>Ragna</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2359706/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hiraki</surname>
<given-names>Jeffrey</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2281695/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xue</surname>
<given-names>Li C.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1309407/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Huynen</surname>
<given-names>Martijn A.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/688281/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Medical Biosciences</institution>, <institution>Radboud University Medical Center</institution>, <addr-line>Nijmegen</addr-line>, <country>Netherlands</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Bio-Prodict</institution>, <addr-line>Nijmegen</addr-line>, <country>Netherlands</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Vartion</institution>, <addr-line>Malden</addr-line>, <country>Netherlands</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/124901/overview">Annalisa Pastore</ext-link>, King&#x2019;s College London, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2309117/overview">Carlos Bueno</ext-link>, Rice University, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2310139/overview">Xinyu Gu</ext-link>, Rice University, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Martijn A. Huynen, <email>martijn.huijnen@radboudumc.nl</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>05</day>
<month>07</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>10</volume>
<elocation-id>1204157</elocation-id>
<history>
<date date-type="received">
<day>11</day>
<month>04</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>06</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Ramakrishnan, Baakman, Heijl, Vroling, van Horck, Hiraki, Xue and Huynen.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Ramakrishnan, Baakman, Heijl, Vroling, van Horck, Hiraki, Xue and Huynen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Predicting pathogenicity of missense variants in molecular diagnostics remains a challenge despite the available wealth of data, such as evolutionary information, and the wealth of tools to integrate that data. We describe DeepRank-Mut, a configurable framework designed to extract and learn from physicochemically relevant features of amino acids surrounding missense variants in 3D space. For each variant, various atomic and residue-level features are extracted from its structural environment, including sequence conservation scores of the surrounding amino acids, and stored in multi-channel 3D voxel grids which are then used to train a 3D convolutional neural network (3D-CNN). The resultant model gives a probabilistic estimate of whether a given input variant is disease-causing or benign. We find that the performance of our 3D-CNN model, on independent test datasets, is comparable to other widely used resources which also combine sequence and structural features. Based on the 10-fold cross-validation experiments, we achieve an average accuracy of 0.77 on the independent test datasets. We discuss the contribution of the variant neighborhood in the model&#x2019;s predictive power, in addition to the impact of individual features on the model&#x2019;s performance. Two key features: evolutionary information of residues in the variant neighborhood and their solvent accessibilities were observed to influence the predictions. We also highlight how predictions are impacted by the underlying disease mechanisms of missense mutations and offer insights into understanding these to improve pathogenicity predictions. Our study presents aspects to take into consideration when adopting deep learning approaches for protein structure-guided pathogenicity predictions.</p>
</abstract>
<kwd-group>
<kwd>protein structure</kwd>
<kwd>3D CNN</kwd>
<kwd>missense variant</kwd>
<kwd>machine learning</kwd>
<kwd>gain-of-function</kwd>
<kwd>loss-of-function</kwd>
</kwd-group>
<contract-sponsor id="cn001">European Regional Development Fund<named-content content-type="fundref-id">10.13039/501100008530</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Radboud Universitair Medisch Centrum<named-content content-type="fundref-id">10.13039/501100006209</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Structural Biology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Numerous Mendelian diseases can be attributed to alterations in the coding regions of the DNA, i.e., missense variants (<xref ref-type="bibr" rid="B27">Kryukov et al., 2007</xref>). With rapid advances in sequencing technologies, the ease and ability to map a person&#x2019;s complete genome has dramatically aided in obtaining genetic diagnosis. Nevertheless, only a small fraction of the missense mutations is pathogenic (<xref ref-type="bibr" rid="B32">Lek et al., 2016</xref>) and for the majority of missense variants it is not clear whether the phenotypic outcome is pathogenic or neutral. Such variants are coined &#x201c;variants of uncertain significance&#x201d; (VUS). Evidently, identifying and comprehending the functional effects of missense variants is of critical importance, not only to understand the etiology of the disease but also towards development of treatment regimens.</p>
<p>Significant advances have been made in the development of variant effect predictors that largely rely on evolutionary conservation, which is a strong signal for predicting pathogenicity. Such evolutionary cues in combination with physicochemical properties of amino acids form the base framework of several state-of-the-art techniques including SIFT (<xref ref-type="bibr" rid="B43">Ng and Henikoff, 2003</xref>), PolyPhen2 (<xref ref-type="bibr" rid="B2">Adzhubei et al., 2010</xref>), CADD (<xref ref-type="bibr" rid="B25">Kircher et al., 2014</xref>), and MutPred (<xref ref-type="bibr" rid="B33">Li et al., 2009</xref>). Although evolutionary information holds value in predicting pathogenicity, it does not provide mechanistic understanding. The mechanisms of the pathogenicity of missense variants are often attributable to perturbations in conformational and functional properties of three-dimensional structures (<xref ref-type="bibr" rid="B68">Wang and Moult, 2001</xref>; <xref ref-type="bibr" rid="B20">Iqbal et al., 2020</xref>), which can contribute to our understanding of the underlying molecular pathology. Several studies have thus incorporated features that leverage structural properties (<xref ref-type="bibr" rid="B65">Venselaar et al., 2010</xref>; <xref ref-type="bibr" rid="B7">Capriotti and Altman, 2011</xref>; <xref ref-type="bibr" rid="B21">Ittisoponpisan et al., 2019</xref>; <xref ref-type="bibr" rid="B31">Laskowski et al., 2020</xref>), protein dynamics (<xref ref-type="bibr" rid="B47">Ponzoni et al., 2020</xref>), protein-protein interaction networks (<xref ref-type="bibr" rid="B69">Yates et al., 2014</xref>), and protein structural stability (<xref ref-type="bibr" rid="B5">Ancien et al., 2018</xref>), to improve pathogenicity predictions on top of what can be achieved with sequence conservations. In the absence of experimental structural information, context-dependent sequence-based models have the potential to accurately capture intra-protein 3D contacts, i.e., via evolutionarily coupled residues (<xref ref-type="bibr" rid="B42">Morcos et al., 2011</xref>; <xref ref-type="bibr" rid="B40">Marks et al., 2012</xref>; <xref ref-type="bibr" rid="B18">Hopf et al., 2014</xref>). Utility of such models has shown reasonable improvement in distinguishing pathogenic missense variants from benign ones (<xref ref-type="bibr" rid="B13">Feinauer and Weigt, 2017</xref>; <xref ref-type="bibr" rid="B19">Hopf et al., 2017</xref>). A complete list of available resources and tools for variant effect prediction and their benchmark evaluation studies has been published elsewhere (<xref ref-type="bibr" rid="B35">Liu et al., 2011</xref>; <xref ref-type="bibr" rid="B38">Livesey and Marsh, 2022</xref>). Despite the significant advances, the challenge of distinguishing pathogenic variants from benign ones remains elusive with most methods exhibiting a wide spectrum of performances on different test datasets (<xref ref-type="bibr" rid="B44">Niroula and Vihinen, 2019</xref>; <xref ref-type="bibr" rid="B37">Livesey and Marsh, 2020</xref>).</p>
<p>Most knowledge-driven approaches that employ machine learning (ML) classifiers rely on various handcrafted features to predict variant effects, which could be time-consuming and laborious. This is compounded by heterogeneity in feature attributes that can pose challenges in data integration (<xref ref-type="bibr" rid="B6">Bagley and Altman, 1995</xref>). Deep learning accelerated approaches can help overcome such limitations. CNNs have gained prominence in the last decade due to their ability to automatically capture patterns from input data as well as the hierarchical representations therein (<xref ref-type="bibr" rid="B26">Krizhevsky et al., 2012</xref>), enabling them to capture relationships between different features. This aspect is particularly useful for analyzing high dimensional data such as protein structures.</p>
<p>Recent efforts have demonstrated the use of 3D-CNNs in exploiting protein structure data for several applications including the prediction of amino acids compatible with protein microenvironments (<xref ref-type="bibr" rid="B61">Torng and Altman, 2017</xref>; <xref ref-type="bibr" rid="B48">Pun et al., 2022</xref>), identification of novel gain-of-function mutations (<xref ref-type="bibr" rid="B58">Shroff et al., 2020</xref>), and the prediction of mutation-induced changes in protein stability (<xref ref-type="bibr" rid="B34">Li et al., 2020</xref>). We introduce DeepRank-Mut, a configurable 3D-CNN framework that predicts pathogenicity of missense variants using wildtype structural microenvironment surrounding the variants in 3D space. The base framework is derived from its parent DeepRank that distinguishes and ranks biologically relevant protein-protein interactions from those that arise due to crystallographic artifacts (<xref ref-type="bibr" rid="B49">Renaud et al., 2021</xref>). The underlying premise of our approach is that the functional outcome of any missense variant is often reflected in the properties of amino acids in the variant neighborhood, in addition to the properties of the variant amino acid itself. Our approach is similar to the method devised by <xref ref-type="bibr" rid="B61">Torng and Altman (2017)</xref>, which, given a site, predicts the amino acids compatible with that specified site based on the surrounding protein microenvironment. In contrast, we train our model explicitly to learn label-specific (benign or pathogenic) features/patterns in the variant neighborhood. Given a missense variant, we first obtain the associated 3D protein structure, either from the protein itself or from a homolog, and calculate features including surface geometry, empirical energies, and atomic densities, in addition to the sequence conservation scores for the mutated site as well as the residues in its neighborhood. These features are mapped onto 3D grids parameterized using properties of the constituent atoms, followed by data augmentation to enrich the input dataset. We then use the power of 3D-CNNs to automatically discern spatially proximal features within these representations.</p>
<p>DeepRank-Mut achieves a performance comparable to techniques that efficiently combine sequence and structure-based features. We analyze the contribution of each of the features to the model&#x2019;s predictive ability, as well as how the neighborhood contributes to the performance. To better understand predictor accuracy, we explore underlying mechanisms of pathogenic mutations and show that the features identify autosomal recessive mutations better than autosomal dominant mutations. We discuss the overall generalizability of our method and provide avenues for better 3D-based missense variant prioritization.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 Datasets</title>
<p>A total of 193,714 missense variants (164,574 benign, 29,140 disease-causing) were collected from ClinVar (<xref ref-type="bibr" rid="B30">Landrum et al., 2018</xref>), gnomAD (<xref ref-type="bibr" rid="B24">Karczewski et al., 2020</xref>) and Dutch genome diagnostic laboratories (<xref ref-type="bibr" rid="B66">VKGL, 2019</xref>), which could be linked to protein structures, either directly or through homology with a sequence identity cut-off of 40%. This cutoff was selected based on previous research that suggests that a 40% identity corresponds to a good likelihood of functional equivalence (<xref ref-type="bibr" rid="B45">Pearson, 2013</xref>). Missense variants were mapped onto protein structures using 3DM systems as a guide (<xref ref-type="bibr" rid="B29">Kuipers et al., 2010</xref>). Independent test datasets were obtained from studies based on BRCA1 (<xref ref-type="bibr" rid="B14">Findlay et al., 2018</xref>), <xref ref-type="bibr" rid="B16">Gunning et al. (2021)</xref> and the InSIGHT database (<xref ref-type="bibr" rid="B60">Thompson et al., 2014</xref>). This resulted in a total of 217,679 missense variants that could reliably be mapped onto 57,551 structures; 25,856 structures were mapped to 40,369 pathogenic variants, and 31,695 structures were mapped to 177,310 benign variants. It should be noted that, at this stage, the structures are mapped regardless of the experimental method used for their determination. Missense variants from ClinVar were incorporated if they had a review status of at least one star, excluding those with conflicting interpretations. &#x201c;Benign&#x201d; and &#x201c;Likely benign&#x201d; ClinVar variants were included and categorized as benign, while &#x201c;Pathogenic&#x201d; and &#x201c;Likely pathogenic&#x201d; variants were incorporated and classified as pathogenic. The gnomAD variants with a minor allele frequency higher than 0.1% were selected and labeled as benign.</p>
<p>Our in-house database, HSSP (<xref ref-type="bibr" rid="B62">Touw et al., 2015</xref>) was consulted to obtain structure-based sequence alignments. Position-specific scoring matrices (PSSMs) were constructed for the alignments using PSI-BLAST (<xref ref-type="bibr" rid="B4">Altschul et al., 1997</xref>) with single iteration. Each of the PSSMs were then mapped back onto their respective structures using the PSSMGen package (<ext-link ext-link-type="uri" xlink:href="https://github.com/DeepRank/PSSMGen">https://github.com/DeepRank/PSSMGen</ext-link>).</p>
</sec>
<sec id="s2-2">
<title>2.2 Data pre-processing</title>
<sec id="s2-2-1">
<title>2.2.1 Feature calculation and voxelization of the neighborhood</title>
<p>We use protein crystal structures of resolution better than 3&#xc5; in our study, as these provide details at the atomic level with high certainty (<xref ref-type="bibr" rid="B70">Zardecki et al., 2022</xref>). Consequently, variants that are mapped to structures solved using methods other than X-ray crystallography, such as NMR or cryo-EM, are excluded. For ease in data handling, we mapped each missense variant to a maximum of three crystal structures of the most similar sequences. For each variant mapped to a crystal structure, we first extract the local neighborhood with a radius of 10&#xc5; around the variant, which typically serves as a distance beyond which the strength of long-range non-bonded interaction energies gradually weakens (<xref ref-type="bibr" rid="B46">Pincus and Scheraga, 1977</xref>). We include residues whose atoms fall within this radius to obtain residue-based features. This is followed by calculation of atomic features such as densities and charges for the wildtype amino acid and the residues in its microenvironment. Pairwise Coulomb and van der Waals potentials are calculated between atoms of the wildtype residue and the residues in the neighborhood. For a given atom, these features are defined as the sum of all pairwise potentials between the atom and its contact atoms. Bonded pairs, i.e., pairs of atoms separated by up to 2 bonds are excluded from this measure. The atomic densities, charges and non-bonded energies are based on the OPLS force field (<xref ref-type="bibr" rid="B22">Jorgensen and Tirado-Rives, 1988</xref>), calculated in the same manner as in the parent DeepRank (<xref ref-type="bibr" rid="B49">Renaud et al., 2021</xref>) (see <xref ref-type="sec" rid="s10">Supplementary Methods</xref>). Solvent accessible surface area (SASA) is calculated using FreeSASA (v2.0.3) (<xref ref-type="bibr" rid="B41">Mitternacht, 2016</xref>). Water molecules in protein structures, when present, are not included in the analysis. In addition to the PSSM obtained for the wildtype and variant amino acids, we also include the PSSM profile for the residues in the variant microenvironment. Such residue-based feature values are assigned to the residue&#x2019;s constituent atoms. All feature values are localized on atoms, to be subsequently mapped on a 3D grid (see <xref ref-type="fig" rid="F1">Figure 1</xref>); only those atoms that lie within 10&#xc5; radius of the variant are considered. At this stage, it should be noted that some structures in the PDB database may contain missing residues that fall within the variant environment radius, leading to errors in the feature mapping step. Such molecules are thus, excluded from the dataset.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>A schematic of the DeepRank-Mut framework. <bold>(A)</bold> The first step includes extraction of the variant environment, where residues within a radius of 10&#xc5; (diameter of 20&#xc5;) around the variant are drawn. As an example case, the crystal structure of phosphoglucomutase (PDB: 1C4G) with the missense variant Asn37 is depicted. This is followed by the feature calculation step where structural properties and PSSM scores are computed for the variant site and the residues in its environment. All features are localized on atoms as illustrated in <bold>(B)</bold>. For simplicity, one structural property (charge), localized on atoms, is shown. A 3D grid of size 20 &#xd7; 20 &#xd7; 20 is centered at the C<italic>&#x3b1;</italic> atom of the residue at variant site, discretized into voxels of 1&#xc5;. <bold>(C)</bold> Each of the features calculated are normalized using standardization and then mapped onto the grid using a Gaussian function. For simplicity, the Gaussian mapping of one feature, i.e., charge for all atoms within a 20&#xc5; box is depicted. In principle, a total of 31 calculated features are mapped to the 3D grid for a given variant. <bold>(D)</bold> This 3D grid with mapped features of shape (20, 20, 20, 31) serves as an input for the 3D-CNN network. The final classification score takes a value between 0 and 1 for each class (benign and pathogenic).</p>
</caption>
<graphic xlink:href="fmolb-10-1204157-g001.tif"/>
</fig>
<p>We construct a 3D grid of size 20&#x00c5; &#xd7; 20&#x00c5; &#xd7; 20&#x00c5; centered at the C<italic>&#x3b1;</italic> atom of the amino acid at the variant site. This 20&#xc5; box is divided into voxels of 1&#xc5;, parameterized with 31 physicochemical property channels (<xref ref-type="table" rid="T1">Table 1</xref>). The properties are mapped on a 3D grid using Gaussian functions to approximate atom connectivity, as demonstrated previously in the parent DeepRank framework (<xref ref-type="bibr" rid="B49">Renaud et al., 2021</xref>). The contribution (<italic>w</italic>
<sub>k</sub>) of an atom <italic>k</italic> to a given grid point is determined based on Gaussian distance dependence, i.e., the contribution decreases with increasing distance between the atom and the grid point. This is given by the equation:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>v</italic>
<sub>
<italic>k</italic>
</sub> is the feature value, <italic>r</italic> denotes position of the grid point and <italic>r</italic>
<sub>
<italic>k</italic>
</sub> denotes atomic coordinates (x, y, z). The standard deviation <italic>&#x3c3;</italic> denotes the van der Waals radius of the associated atom.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>List of features calculated for the residue at the mutation site and the residues in its neighborhood.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Features</th>
<th align="left">Number of channels</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Atomic densities (C, N, O, S)</td>
<td align="left">4</td>
</tr>
<tr>
<td align="left">Atomic charges</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">Solvent accessibility</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">Coulomb potential</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">van der Waals potential</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">Wildtype score: PSSM</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">Variant score: PSSM</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">Information content (PSSM)</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">PSSM profile</td>
<td align="left">20</td>
</tr>
<tr>
<td align="left">Total</td>
<td align="left">31</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>All features, including residue-level features such as sequence conservation scores, are localized on atoms. The two sequence-based features (wildtype and variant probability) are mapped to the atoms of a given wildtype residue.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The feature maps are stacked to create a tensor of shape (20, 20, 20, 31) that then serves as an input to the neural network. We also normalize features of the input data using standardization prior to training. To optimize for speed and efficient handling of large volumes of data, we developed a distributed data preprocessing framework with GPU support, which enabled faster preprocessing times and scalability during numerous iterations of experiments (see <xref ref-type="sec" rid="s10">Supplementary Methods</xref>, <xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>).</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Data augmentation</title>
<p>Prior to the training step, we enrich each of the input 3D grids using data augmentation where a given grid is randomly rotated around its center, and features are mapped onto the grid subsequently. Such a strategy has been shown to improve the performance of CNNs (<xref ref-type="bibr" rid="B57">Shorten and Khoshgoftaar, 2019</xref>). For the current study, we used 5 augmentations based on hyper parameter tuning experiments (<xref ref-type="sec" rid="s10">Supplementary Figure S3</xref>). We did not experiment with a higher number of augmentations due to the infeasible computational costs involved.</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Network architecture</title>
<p>The network used in our study includes a sequential organization of three 3D convolutional layers, alternating with one 3D max pooling layer followed by two fully connected layers (<xref ref-type="fig" rid="F1">Figure 1</xref>). We include batch normalization layers, in addition to dropout layers between the fully connected layers to regularize the model. Details of the architecture are provided in <xref ref-type="table" rid="T2">Table 2</xref> and the complete schema is provided in <xref ref-type="sec" rid="s10">Supplementary Figure S2</xref>. Each 3D convolution layer comprises a set of learnable filters that traverse the input space (depth, height and width) with a stride of 1, capturing local spatial patterns in the variant environment. The output from convolution operations, i.e., the computed feature maps are transformed by a rectified linear activation function (ReLU), which allows the network to identify and extract meaningful spatial features. This is followed by dimension reduction using max pooling operation and a final 3D convolutional layer with ReLU. The transformed output is then flattened to a one-dimensional vector that serves as an input to two fully connected layers. The two final layers integrate the features and apply a set of weights that are optimized during the training step to map extracted features to target classes. The output is then passed through the softmax function which provides the final classification score, a probability estimate between 0 and 1, each for benign and pathogenic classes.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Network architecture used in DeepRank-Mut.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Layer</th>
<th align="left">Size</th>
<th align="left">Output shape</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Batch normalization layer 1</td>
<td align="left">Input</td>
<td align="left">20 &#xd7; 20 &#xd7; 20 &#xd7; 31</td>
</tr>
<tr>
<td align="left">3D convolutional layer 1</td>
<td align="left">20 &#xd7; 20 &#xd7; 20, 31 filters, kernel size &#x3d; 2, stride &#x3d; 1</td>
<td align="left">19 &#xd7; 19 &#xd7; 19 &#xd7; 31</td>
</tr>
<tr>
<td align="left">Batch normalization layer 2</td>
<td align="left"/>
<td align="left">19 &#xd7; 19 &#xd7; 19 &#xd7; 31</td>
</tr>
<tr>
<td align="left">3D convolutional layer 2</td>
<td align="left">19 &#xd7; 19 &#xd7; 19, 64 filters, kernel size &#x3d; 2, stride &#x3d; 1</td>
<td align="left">18 &#xd7; 18 &#xd7; 18 &#xd7; 64</td>
</tr>
<tr>
<td align="left">Batch normalization layer 3</td>
<td align="left"/>
<td align="left">18 &#xd7; 18 &#xd7; 18 &#xd7; 64</td>
</tr>
<tr>
<td align="left">3D max pooling layer</td>
<td align="left">Stride &#x3d; 2</td>
<td align="left">9 &#xd7; 9 &#xd7; 9 &#xd7; 64</td>
</tr>
<tr>
<td align="left">3D convolutional layer 3</td>
<td align="left">9 &#xd7; 9 &#xd7; 9, 64 filters, kernel size &#x3d; 3, stride &#x3d; 1</td>
<td align="left">7 &#xd7; 7 &#xd7; 7 &#xd7; 64</td>
</tr>
<tr>
<td align="left">Batch normalization layer 4</td>
<td align="left"/>
<td align="left">7 &#xd7; 7 &#xd7; 7 &#xd7; 64</td>
</tr>
<tr>
<td align="left">Flatten</td>
<td align="left">7 &#xd7; 7 &#xd7; 7 &#xd7; 64</td>
<td align="left">21,952</td>
</tr>
<tr>
<td align="left">Batch normalization layer 5</td>
<td align="left"/>
<td align="left">21,952</td>
</tr>
<tr>
<td align="left">Fully connected layer 1</td>
<td align="left">21,952 &#xd7; 100 neurons</td>
<td align="left">100 neurons</td>
</tr>
<tr>
<td align="left">Dropout (<italic>p</italic> &#x3d; 0.5)</td>
<td align="left"/>
<td align="left"/>
</tr>
<tr>
<td align="left">Fully connected layer 2</td>
<td align="left">100 &#xd7; 100 neurons</td>
<td align="left">100 neurons</td>
</tr>
<tr>
<td align="left">Dropout (<italic>p</italic> &#x3d; 0.5)</td>
<td align="left"/>
<td align="left"/>
</tr>
<tr>
<td align="left">Softmax</td>
<td align="left">100 &#xd7; 2</td>
<td align="left">2 scores (benign, pathogenic)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-4">
<title>2.4 Training</title>
<p>We performed 10-fold cross validation experiments while ensuring that the missense variants in the training and validation sets are from different proteins, to avoid type 1 circularity in predictions (<xref ref-type="bibr" rid="B17">Heijl et al., 2020</xref>). The test dataset included missense variants independent from the 10-fold training and validation sets. Most genetic variation is neutral, and it is therefore rather common to observe a higher number of benign variants than pathogenic variants in the training data, which has the potential to bias training and performance. We thus constructed balanced subsets of randomly sampled benign and pathogenic missense variants for each of the 10-fold runs. For efficient memory handling, we employed training in mini-batches of 256 variant instances which amounted to &#x223c;1,200 mini-batches per epoch. An epoch refers to a single pass through the complete training data during which the model weights are adjusted to minimize the error between predicted and true label for each input. With the input dataset, one epoch in our approach referred to one pass through more than 280,000 variant instances. We used the AdamW optimizer (<xref ref-type="bibr" rid="B39">Loshchilov and Hutter, 2019</xref>) with a learning rate of 0.001 and weight decay of 0.005 to train our model for 10 epochs. We used cross entropy loss during training, which attempts to minimize the differences in probability distributions between predicted and ground truth labels by adjusting weights. A dropout rate of 0.5 was used to regularize the model. The hyperparameters including number of convolutional layers, number of max pooling layers, grid size, were optimized based on performance on validation set across 10 folds, starting from default parameters of the parent DeepRank.</p>
</sec>
<sec id="s2-5">
<title>2.5 Evaluation metrics</title>
<p>Two metrics, Matthews Correlation Coefficient (MCC) and accuracy, were used to evaluate the performance of DeepRank-Mut. The primary metric used was MCC, as it offers a reliable statistical measure by taking all four categories-true positives (TP), true negatives (TN), false positives (FP), and false negatives (FN) into account, proportional to the size of the binary classes (Eq. <xref ref-type="disp-formula" rid="e2">2</xref>). The usefulness of MCC over accuracy or F1 scores for binary classification has been demonstrated previously (<xref ref-type="bibr" rid="B10">Chicco and Jurman, 2020</xref>).<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>For comparative evaluation with popular state-of-the-art variant effect predictors, we used precomputed pathogenicity prediction scores of 8 algorithms from dbNSFP v4.3 database (<xref ref-type="bibr" rid="B36">Liu et al., 2020</xref>; <xref ref-type="bibr" rid="B35">2011</xref>), including SIFT4G (<xref ref-type="bibr" rid="B63">Vaser et al., 2016</xref>), PolyPhen2 (<xref ref-type="bibr" rid="B2">Adzhubei et al., 2010</xref>), MutationTaster (<xref ref-type="bibr" rid="B54">Schwarz et al., 2014</xref>), MutationAssessor (<xref ref-type="bibr" rid="B50">Reva et al., 2011</xref>), FATHMM (<xref ref-type="bibr" rid="B56">Shihab et al., 2013</xref>), VEST4 (<xref ref-type="bibr" rid="B8">Carter et al., 2013</xref>), PROVEAN (<xref ref-type="bibr" rid="B11">Choi and Chan, 2015</xref>) and MutPred (<xref ref-type="bibr" rid="B33">Li et al., 2009</xref>), as well as prediction scores from Helix (<xref ref-type="bibr" rid="B67">Vroling and Heijl, 2021</xref>). Where available, we used &#x201c;converted rankscores&#x201d; from dbNSFP to ensure that a higher score always indicated higher likelihood of pathogenicity. We excluded meta predictors from this comparison, as well as those that combine annotations from other tools, to account for methods that rely on first principles to predict functional effects of missense variants.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Overview of the datasets and DeepRank-Mut</title>
<p>Training, validation, and test sets are often generated using a simple random split. However, this can result in over fitting and misleading results due to data leakage between the training and evaluation sets (<xref ref-type="bibr" rid="B17">Heijl et al., 2020</xref>). Data splitting at the level of proteins or genes, where training sets never include any data samples from proteins that occur in the validation or test set is used to mitigate this. We split our dataset into 10 pairs of training and test sets, each containing 90% and 10% of the full dataset, respectively, allowing 10-fold cross validation on the full dataset. Independent test sets were gathered from three studies, as described in methods, to aid in the final assessment of the tool&#x2019;s performance. These test sets have been selected as they not only cover genes in-depth (<xref ref-type="bibr" rid="B60">Thompson et al., 2014</xref>; <xref ref-type="bibr" rid="B14">Findlay et al., 2018</xref>) but are also aimed at benchmarking pathogenicity predictors specifically (<xref ref-type="bibr" rid="B16">Gunning et al., 2021</xref>).</p>
<p>After splitting the data, the balanced subsets of randomly sampled benign and pathogenic variants, each mapped to at most three structures, comprised a total of &#x223c;50,000 instances in the training set, &#x223c;4,700 instances in validation and 6,571 in the test set, per fold. The test set was kept identical across all cross-validation folds for an unbiased evaluation of the model.</p>
<p>DeepRank-Mut retains its modularity in implementing data pre-processing steps and training the deep neural network, similar to its parent DeepRank (<xref ref-type="bibr" rid="B49">Renaud et al., 2021</xref>). It allows for flexibility in tasks including feature calculations, setting the grid size and grid resolution, data augmentation, as well as optimizing hyperparameters of the neural network. The base requirements of DeepRank-Mut include a dataset of variants with labels (benign or pathogenic), a dataset of variant-structure maps where each variant is linked to a 3D structure (either experimentally determined or evolutionarily related), a dataset of 3D structures and an optional dataset of PSSM profiles derived for each structure. As detailed in the methods, the framework computes physicochemical properties of the amino acid at the variant site as well as its environment within a radius of 10&#xc5;, followed by voxelization to encode the atomic neighborhood of residues (<xref ref-type="fig" rid="F1">Figures 1A, B</xref>). Our approach relies on leveraging local properties of sites characteristic of benign or pathogenic variants, as pathogenic variants generally tend to occur in regions important for structural/functional integrity of the protein (<xref ref-type="bibr" rid="B20">Iqbal et al., 2020</xref>), like its hydrophobic core. We thus compute a total of 31 features (<xref ref-type="table" rid="T1">Table 1</xref>), encompassing structural and sequence-based properties, for the residue at the variant site and residues spatially proximal to it. The computed features are mapped to a 3D grid where each voxel is parameterized with the feature channels (<xref ref-type="fig" rid="F1">Figure 1C</xref>), which is then followed by data augmentation. As a given variant environment can differ in orientation within or across proteins, the data augmentation step accounts for rotational invariance, thereby improving the model&#x2019;s robustness to variations in input data (<xref ref-type="sec" rid="s10">Supplementary Figure S3</xref>). From our dataset of structures and missense variants, we generated &#x223c;300,000 augmented grids per fold dataset, which were used as input to 3D-CNN (<xref ref-type="fig" rid="F1">Figure 1D</xref>). Each augmented 3D grid is treated as a separate variant instance, thus our model outputs 6 predictions per missense variant (origin grid &#x2b;5 augmented grids) which are averaged to give one final classification score.</p>
</sec>
<sec id="s3-2">
<title>3.2 Overall performance</title>
<p>Our approach achieved a mean accuracy of 0.77 and an average MCC score of 0.52 across the test datasets, with an average sensitivity (true positive rate) of 0.75 and an average specificity (true negative rate) of 0.78 (<xref ref-type="fig" rid="F2">Figure 2</xref>; <xref ref-type="table" rid="T3">Table 3</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Overall performance of DeepRank-Mut. <bold>(A)</bold> The performance metrics of DeepRank-Mut on test sets across 10 folds are depicted as boxplots for true positive rate (TPR), false positive rate (FPR), accuracy and MCC. <bold>(B)</bold> Confusion matrix depicting the average of TP, FP, FN, and TN across 10 folds.</p>
</caption>
<graphic xlink:href="fmolb-10-1204157-g002.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Details of the performance metrics of DeepRank-Mut on test sets across 10 folds.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Fold</th>
<th align="left">TPR</th>
<th align="left">FPR</th>
<th align="left">Accuracy</th>
<th align="left">MCC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">1</td>
<td align="left">0.613</td>
<td align="left">0.142</td>
<td align="left">0.78</td>
<td align="left">0.49</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">0.765</td>
<td align="left">0.241</td>
<td align="left">0.76</td>
<td align="left">0.52</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">0.70</td>
<td align="left">0.124</td>
<td align="left">0.79</td>
<td align="left">0.59</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">0.768</td>
<td align="left">0.258</td>
<td align="left">0.76</td>
<td align="left">0.51</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">0.777</td>
<td align="left">0.24</td>
<td align="left">0.77</td>
<td align="left">0.52</td>
</tr>
<tr>
<td align="left">6</td>
<td align="left">0.787</td>
<td align="left">0.188</td>
<td align="left">0.80</td>
<td align="left">0.60</td>
</tr>
<tr>
<td align="left">7</td>
<td align="left">0.713</td>
<td align="left">0.235</td>
<td align="left">0.74</td>
<td align="left">0.48</td>
</tr>
<tr>
<td align="left">8</td>
<td align="left">0.777</td>
<td align="left">0.191</td>
<td align="left">0.79</td>
<td align="left">0.58</td>
</tr>
<tr>
<td align="left">9</td>
<td align="left">0.758</td>
<td align="left">0.326</td>
<td align="left">0.72</td>
<td align="left">0.43</td>
</tr>
<tr>
<td align="left">10</td>
<td align="left">0.716</td>
<td align="left">0.186</td>
<td align="left">0.76</td>
<td align="left">0.53</td>
</tr>
<tr>
<td align="left">Average</td>
<td align="left">0.737</td>
<td align="left">0.213</td>
<td align="left">0.77</td>
<td align="left">0.52</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3-2-1">
<title>3.2.1 Impact of individual features and the variant environment on the performance</title>
<p>To investigate the contribution of neighborhood in the predictor accuracies, we compared the performance of our 3D-CNN model trained on all features to those trained separately on-a) PSSM features, b) structural features, c) variant site-specific PSSMs (<xref ref-type="fig" rid="F3">Figure 3A</xref>). The model trained on PSSM features included PSSMs for the residues in the 3D neighborhood as well as the scores for wildtype and variant amino acids, while the model trained on variant site-specific PSSMs was devoid of the neighborhood profile. As illustrated in the figure, the features derived from the neighborhood, in the 3D context, seemingly hold more information than the site-specific features. This aspect was also observed during hyperparameter tuning experiments, where a range of different sizes of 3D grids were tested to find the optimal grid size. Models with smaller variant neighborhoods (grid sizes &#x3d; 7&#xc5;, 8&#xc5;) performed poorly on validation sets as compared to the models with grid size of 15&#xc5; and 20&#xc5; (<xref ref-type="sec" rid="s10">Supplementary Figure S4</xref>). It has been reported earlier that the atomic details do not provide significant information for local protein environments beyond a 20&#xc5; cutoff (<xref ref-type="bibr" rid="B6">Bagley and Altman, 1995</xref>). An optimal grid size of 20&#xc5; was thus chosen for all experiments. Additionally, we investigated the apparent contribution of individual structural features in prediction accuracies, as illustrated in <xref ref-type="fig" rid="F3">Figure 3B</xref>. We note that solvent accessibility of residues has the most predictive capacity amongst all structural features. Residues buried in the hydrophobic core of the protein are often associated with pathogenicity, while solvent-exposed missense variants are often found to be enriched in populations, as also exemplified by <xref ref-type="bibr" rid="B20">Iqbal et al. (2020)</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Contribution of the variant neighborhood and features to the model&#x2019;s predictive ability. <bold>(A)</bold> Receiver operating characteristic (ROC) curves are drawn from scores generated by the complete model (red) and by models separately trained only on the PSSM profile of residues in the neighborhood (green), only on the structural features from the neighborhood (cyan) and PSSM of mutation site alone (purple). The AUC values obtained illustrate the value of using 3D neighborhood information in the predictions. <bold>(B)</bold> ROC curves are drawn using scores generated by models separately trained on the individual structural features. <bold>(C)</bold> ROC curves for leave-one-feature-out analysis are drawn using scores from models trained without a specific feature. Model trained without PSSMs (beige) and the model trained on structural features from neighborhood (cyan) in the first panel <bold>(A)</bold> are identical. <bold>(D)</bold> ROC curves are drawn for scores from models where redundant features are removed. For consistency the model trained on all 31 features is included in all panels. Total <italic>n</italic> &#x3d; 6,571 instances, 3,804 pathogenic and 2,767 benign.</p>
</caption>
<graphic xlink:href="fmolb-10-1204157-g003.tif"/>
</fig>
<p>Additionally, we also performed leave-one-feature-out analysis to assess redundancy in our feature selection. <xref ref-type="fig" rid="F3">Figure 3C</xref> illustrates similarity in ROC curves of models trained without pairwise potentials (Coulomb &#x2b; van der Waals), atomic charges and atomic densities. The contributions of these features in prediction accuracies are similar as also noted in <xref ref-type="fig" rid="F3">Figure 3B</xref>, suggesting redundancies in features employed. Subsequently, we tested our model&#x2019;s performance by excluding seemingly redundant features, such as atomic densities and charges from the feature set (<xref ref-type="fig" rid="F3">Figure 3D</xref>). Although minimal, the contribution of each of the structural features holds value in the overall performance. Significantly, solvent accessibility and PSSMs show considerable impact on the model&#x2019;s performance.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Comparison with state-of-the-art resources</title>
<p>We used precomputed pathogenicity scores of 8 algorithms from dbNSFP database as well as scores from the Helix for the test dataset used in the study. In the case of PolyPhen2, we used scores from the HumVar-trained models as recommended by the authors for the purpose of distinguishing variants with drastic functional effects from benign ones (<xref ref-type="bibr" rid="B3">Adzhubei et al., 2013</xref>). <xref ref-type="fig" rid="F4">Figure 4</xref> illustrates the ROC curves drawn from these scores along with those from DeepRank-Mut for the variant predictions available for each algorithm. While the performance of our approach is seemingly comparable to other widely-used resources that incorporate sequence conservation and structural features, such as MutPred (<xref ref-type="bibr" rid="B33">Li et al., 2009</xref>) and PolyPhen2 (<xref ref-type="bibr" rid="B2">Adzhubei et al., 2010</xref>), it must be noted that the available variant predictions for these tools constitute 62% and 72% of the total test set, respectively (<italic>n</italic> in <xref ref-type="fig" rid="F4">Figure 4</xref>, <xref ref-type="sec" rid="s10">Supplementary Table S1</xref>). Both these ML-based tools incorporate several handcrafted features, aside from sequence conservation, including secondary structural assignments, normalized B-factors, and various annotations of functional sites; the only overlapping features with DeepRank-Mut being SASA and sequence conservation. Helix, built on proprietary structure-based sequence alignments (<xref ref-type="bibr" rid="B29">Kuipers et al., 2010</xref>; <xref ref-type="bibr" rid="B67">Vroling and Heijl, 2021</xref>), and VEST4, a variant prioritization tool that explores enrichment of functional variants across disease exomes (<xref ref-type="bibr" rid="B8">Carter et al., 2013</xref>), were notably the top performers.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Comparison with other state-of-the-art resources. ROC curves drawn from scores generated by various pathogenicity predictors, including DeepRank-Mut, are shown based on the test variants available for each predictor in dbNSFP.</p>
</caption>
<graphic xlink:href="fmolb-10-1204157-g004.tif"/>
</fig>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 3D-CNNs appear less powered to identify outcome of solvent-exposed variants</title>
<p>We examined our model&#x2019;s predictive ability by analyzing missense variants in the test-set that were consistently predicted incorrectly across all 10 folds. We explored the aspects that promoted incorrect classification. A total of 2,883 missense variants were found to be incorrectly classified across the cross-validation experiments, of which more than half (1,732) consisted of misclassified pathogenic variants. We computed relative solvent accessibilities (RSA) for each variant residue, by dividing their absolute solvent accessibilities in &#xc5;<sup>2</sup> by their maximum allowed solvent accessibilities obtained from Rost and Sander (<xref ref-type="bibr" rid="B52">Rost and Sander, 1994</xref>). Residues were categorized as solvent-exposed if the RSA values were &#x3e;20% and buried if below 20%. Using these a substantial proportion of the misclassified pathogenic variants was found to be solvent-exposed (<xref ref-type="sec" rid="s10">Supplementary Table S2</xref>).</p>
<p>We constructed 2 &#xd7; 2 contingency tables based on the correct and incorrect classifications with respect to solvent accessibility of the associated variants. <xref ref-type="fig" rid="F5">Figure 5</xref> illustrates the role of solvent accessibility in the predicted outcomes. The misclassified variants pertained to solvent-exposed pathogenic variants and buried benign variants (<xref ref-type="fig" rid="F5">Figure 5A</xref>, odds ratio &#x3d; 0.27). That we are relatively successful in predicting pathogenicity in buried variants is consistent with the notion of buried enrichment of pathogenic variants (<xref ref-type="bibr" rid="B20">Iqbal et al., 2020</xref>; <xref ref-type="bibr" rid="B53">Savojardo et al., 2020</xref>). The distribution of raw atom-level solvent accessibility values across benign and pathogenic classes calculated in our approach is illustrated in <xref ref-type="sec" rid="s10">Supplementary Figure S9</xref>. Two reasons for the quality of the predictions could be postulated: a) considering the contribution of SASA in the model&#x2019;s performance, it is likely that the model is unable to generalize on missense variants that fall outside the purview of typical SASA distribution observed in benign and pathogenic variants, or b) the 3D input grids for solvent-exposed missense variants are sparsely populated which leads to a lack of discernible patterns/features for the model to learn from.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Association of solvent accessibility of variants in prediction outcomes. <bold>(A)</bold> Bar charts for correctly classified and misclassified variants with respect to their solvent accessibility are shown. <bold>(B)</bold> The performance metrics on test data in terms of TP, FP, FN, TN are depicted as bar charts for models trained on all variants (full model), on only buried variants (buried model) and on solvent-exposed variants alone (solvent-exposed model). The proportion of true positives, i.e., pathogenic variants in the model trained on buried variants is notably high.</p>
</caption>
<graphic xlink:href="fmolb-10-1204157-g005.tif"/>
</fig>
<p>We created separate training subsets of buried and solvent-exposed variants to understand 3D-CNN&#x2019;s generalizability to either subset. We observed that the predictions on pathogenic variants improved with the model trained on buried missense variants alone, however, this model misclassified much of the benign variants, whereas the model trained on solvent-exposed variants alone showed a performance comparable to that of the full model trained on all variants (<xref ref-type="fig" rid="F5">Figure 5B</xref>; <xref ref-type="sec" rid="s10">Supplementary Figure S5</xref>). It is possible that the presence of a large proportion of solvent-exposed variants in our training data may have impacted the performance (<xref ref-type="sec" rid="s10">Supplementary Figure S9</xref>). Furthermore, to assess whether sparsity of 3D grids of solvent-exposed variants affected the model&#x2019;s performance, we calculated the ratio of solvent (void) voxels to atom-contained (non-void) voxels in the 3D grids in test dataset and compared the distribution of these ratios against the corresponding pathogenic and benign prediction scores. We find no correlation for pathogenic variants (Pearson&#x2019;s <italic>r</italic> &#x3d; &#x2212;0.09), while we find that the presence of void voxels is weakly indicative of correct classifications for benign variants (Pearson&#x2019;s <italic>r</italic> &#x3d; &#x2212;0.24) (<xref ref-type="sec" rid="s10">Supplementary Figure S10</xref>). This overall suggests that grid sparsity has weak effect on the correct classification of benign variants, whereas the incorrect classifications of solvent-exposed pathogenic variants is possibly due to other reasons, such as lack of function-specific features, and/or incomplete knowledge of their interaction partners.</p>
<p>Since data augmentation and feature normalization strategies, typically used to circumvent lack of generalizability and potential biases, are already incorporated in our approach we experimented with inclusion of other structural features: secondary structural content and normalized B-factors. The premise behind use of secondary structural content was based on the report by Abrus&#xe1;n and Marsh (<xref ref-type="bibr" rid="B1">Abrus&#xe1;n and Marsh, 2016</xref>), who showed differences in the ability of alpha helices and beta strands to tolerate mutations. Secondary structural assignments for protein structures were obtained from our in-house database (DSSP v.3.1.4) (<xref ref-type="bibr" rid="B23">Kabsch and Sander, 1983</xref>), and were stored as one-hot encoded features in 3D grids. B-factors or temperature factors are obtained from X-ray crystallography experiments that indicate atomic flexibility in the protein&#x2019;s crystalline state, and are known to correlate with flexible regions of the protein. Based on the earlier reports of active/functional sites associated with lower B-factors as compared to non-functional residues (<xref ref-type="bibr" rid="B59">Sun et al., 2019</xref>), we used normalized B-factors as a feature to potentially capture such differences. However, the two additional features did not serve as strong determinants of pathogenicity (<xref ref-type="sec" rid="s10">Supplementary Figure S6</xref>). The relatively low quality of predictions for solvent-exposed pathogenic variants and buried benign variants could be due to lack of function-specific features.</p>
</sec>
<sec id="s3-4">
<title>3.4 Success of pathogenicity prediction depends on underlying disease mechanisms</title>
<p>We further investigated DeepRank-Mut&#x2019;s generalizability with respect to mutation mechanisms. Most available pathogenicity predictors do not make a distinction between different types of mutation mechanisms such as loss-of-function (LoF) or gain-of-function (GoF), that are often linked to mode of inheritance. LoFs are function-disrupting mutations that usually cause damage to protein structures and are straightforward to comprehend and identify, as they are generally not tolerated at sites of high structural and/or functional importance, and lead to degradation of the protein. In contrast, GoFs exhibit milder effects on protein stability while giving rise to altered protein functions that lead to diseases (<xref ref-type="bibr" rid="B15">Gerasimavicius et al., 2022</xref>). In terms of mode of inheritance, autosomal recessive (AR) diseases are predominantly linked to LoFs, while autosomal dominant (AD) diseases manifest through mechanisms such as GoFs, dominant-negative mutations (DN) as well as through LoFs, i.e., haploin sufficiency (<xref ref-type="bibr" rid="B64">Veitia et al., 2018</xref>).</p>
<p>To understand how DeepRank-Mut generalizes on distinct modes of inheritance of pathogenic variants, we split our test datasets into variants with AD inheritance (<italic>n</italic> &#x3d; 1,363; 550 benign, 813 pathogenic) and variants with AR inheritance (<italic>n</italic> &#x3d; 563; 244 benign, 319 pathogenic), based on information obtained from ClinVar (<xref ref-type="bibr" rid="B30">Landrum et al., 2018</xref>). Only a smaller subset could be mapped to crystal structures: 585 structures mapped to 515 AD variants, and 77 structures mapped to 132 variants. We did not filter the AD dataset further to segregate mutations into haploinsufficient genes (LoFs) and non-LoFs (GoFs, DNs), due to lack of detailed annotations of non-LoFs in ClinVar. However, it is worth noting that mutations in the AD dataset could consist of higher proportion of LoFs than non-LoFs due to smaller mutational target for non-LoFs, i.e., fewer mutations alter protein function than disrupt it. <xref ref-type="fig" rid="F6">Figure 6</xref> illustrates a marked difference in the model&#x2019;s performance between the two datasets, suggesting dependence on underlying effects of the variant on the protein. It is apparent from the figure that our model is able to generalize AR mutations (LoFs) better than AD mutations (LoFs and non-LoFs). Details on the pathogenicity predictions obtained for AD and AR datasets, are provided in <xref ref-type="sec" rid="s10">Supplementary Table S3</xref>. To further examine our relative success in correctly classifying buried pathogenic variants and AR variants we analyzed the distribution of solvent accessibility in the AD and AR datasets. Interestingly, the typical distribution of solvent-exposed benign variants and buried pathogenic variants was found to be more pronounced in AR datasets than in the AD datasets (<xref ref-type="fig" rid="F7">Figure 7</xref>), explaining the relative success of our model in distinguishing LoF mutations from benign (<xref ref-type="sec" rid="s10">Supplementary Figure S7</xref>, MCC &#x3d; 0.67).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Impact of underlying disease mechanisms on pathogenicity predictions. Performance of DeepRank-Mut on two datasets that are divided based on mode of inheritance. ROC curves are drawn for scores generated from the model tested on variants with AD inheritance (<italic>n</italic> &#x3d; 585), and from the model tested on variants with AR inheritance (<italic>n</italic> &#x3d; 77). The AUC values are markedly different between the two datasets as depicted. It must be noted that the predictions are made for those variants that could be mapped to protein crystal structures.</p>
</caption>
<graphic xlink:href="fmolb-10-1204157-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Association of missense variants across predictions on different test datasets with solvent accessibility. The bar plot shows the proportion of surface-exposed and buried missense variants in each of the binary outcomes for each of the datasets. &#x201c;All&#x201d; denotes all input variants, AD denotes mutations with autosomal dominant inheritance, and AR denotes mutations with AR inheritance. The log-odds ratio is calculated for each case to determine the strength of association between the binary feature (buried or surface-exposed) and the binary outcome (benign or pathogenic).</p>
</caption>
<graphic xlink:href="fmolb-10-1204157-g007.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>Numerous efforts in the last decade have aided in the general understanding of effects of disease-causing mutations on the biophysical characteristics of proteins-including protein stability, dynamics, and protein-protein interactions (<xref ref-type="bibr" rid="B28">Kucukkal et al., 2015</xref>; <xref ref-type="bibr" rid="B20">Iqbal et al., 2020</xref>). It has been observed that pathogenic mutations are often associated with changes in local hydrogen-bonding network, electrostatic interactions, and overall side-chain geometry (<xref ref-type="bibr" rid="B28">Kucukkal et al., 2015</xref>). Although this knowledge has helped in the advancement of variant effect predictors that integrate various information on top of sequence-based features, the accurate prediction of a functional outcome of a missense variant is often fraught with challenges that we partly bring forth in this study.</p>
<p>We describe DeepRank-Mut, a structure-guided approach that leverages properties in the local variant neighborhood and uses 3D-CNNs to draw relationships between the spatially proximal features to distinguish pathogenic missense variants from benign. Our approach is robust to rotational variations, as we account for different orientations of a given variant environment through data augmentation steps. We did not experiment with larger augmentations due to large computational costs incurred. The performance of DeepRank-Mut was found to be comparable with other widely used predictors, such as PolyPhen2 which employs classical ML algorithm and relies on handcrafted features. Our investigations into the generalizability of our model revealed aspects that could be of interest to those who adopt deep learning techniques in structure-based variant effect predictions.</p>
<p>We find that the evolutionary information (PSSM profile) of the variant neighborhood captures patterns in the 3D structural context of variant sites better than the individual structural properties themselves. In contrast, inclusion of variant site-specific conservation scores alone, devoid of the 3D context, render the 3D-CNN model myopic thereby affecting the overall predictive ability. This finding is of considerable significance as it shows that the model potentially draws context dependence in terms of evolutionarily coupled residues. Pairs of residues under structural and functional constraints can exhibit strong inter-residue correlations, and thus coevolve (<xref ref-type="bibr" rid="B12">de Juan et al., 2013</xref>). Such a property has been shown to be useful in capturing effects of genetic variations (<xref ref-type="bibr" rid="B19">Hopf et al., 2017</xref>). Without explicitly modeling such inter-residue correlations, the performance of our model trained only on the PSSM profile of the neighborhood illustrates the utility of 3D-CNNs in capturing complex relationships between residues. This is further strengthened by the leave-one-feature-out analysis, where exclusion of seemingly redundant features from the model affected its performance.</p>
<p>Solvent accessible surface area was identified as the second most important feature that contributed to the predictor accuracies. Considering earlier reports on the enrichment of solvent-exposed missense variants in populations and enrichment of pathogenic variants in the hydrophobic core of proteins (<xref ref-type="bibr" rid="B20">Iqbal et al., 2020</xref>; <xref ref-type="bibr" rid="B53">Savojardo et al., 2020</xref>), we sought to explore their distribution in missense variants which were consistently misclassified across our datasets. We note that a significant proportion of misclassified pathogenic variants were found to be solvent-exposed, which raises the question whether our model loses generalizability while prioritizing buried pathogenic variants. Our experiments with models separately trained on buried and solvent-exposed missense variants yielded interesting results. The buried model could correctly identify pathogenic variants, even those that are solvent-exposed, while misclassifying a significant proportion of benign variants. The solvent-exposed model, on the other hand, showed similar performance in comparison to the original full model trained on all variants. These findings necessitate incorporating function-specific features or use of other suitable representations of protein structures, such as graphs, to adequately capture the underlying differences within pathogenic missense variants. Achieving high classification scores on solvent-exposed variants do pose a challenge, yet may be overcome with the following strategies: a) ensemble learning, combining multiple models trained on different feature sets related to solvent-exposed variants, such as ligand binding sites or phosphorylation sites; b) active learning, iteratively selecting the most informative solvent-exposed variants for labeling and training the model; or c) self-supervised learning, training the model to predict masked residues. Moreover, it is also possible that the solvent-exposed pathogenic variant site is a part of a larger assembly or participates in protein-protein interactions, an aspect not considered in this study. Use of full protein complex structures for pathogenic variants, wherever applicable, or features that indicate their role in function could help improve classifications (<xref ref-type="bibr" rid="B15">Gerasimavicius et al., 2022</xref>). Overall, we find that the two main features: evolutionary information of residues in the variant neighborhood and solvent accessibilities sufficiently capture most of the important traits around variant sites.</p>
<p>Consideration of disease mechanisms appears to be crucial in the quality of pathogenicity predictions, as exemplified in our study. Our approach could generalize on mutations linked to AR inheritance better than the mutations linked to AD inheritance, corroborating results from an earlier study by <xref ref-type="bibr" rid="B15">Gerasimavicius et al. (2022)</xref>. This finding is primarily due to the underlying mechanisms of mutations where protein destabilizing LoFs, often associated with AR diseases, are more straightforward to identify than non-LoFs which tend to have milder impacts on protein stability. Moreover, distribution of solvent accessibility of variants was suggestive of notable differences in the proportion of buried and solvent-exposed pathogenic variants, across the datasets. The overall performance of AR datasets over AD dataset is potentially due to two plausible reasons: a) feature representations are sufficiently able to distinguish LoFs from benign, and not non-LoFs from benign and b) limited amount of data on variants with non-LoF mechanisms. Both these postulates hold true considering the damaging effects on protein structure caused by LoFs that are relatively straightforward to discern (<xref ref-type="bibr" rid="B15">Gerasimavicius et al., 2022</xref>), and considering the total size of missense variants with non-LoF mechanisms (GoF and DN) mapped onto protein structures (<italic>n</italic> &#x3d; 972), which is insufficient for training using deep neural networks. Since we did not segregate the AD dataset further into non-LoFs (GoFs, DNs) and LoFs, i.e., mutations in haploinsufficient genes, it is not apparent how the PSSM profile of residues in a variant environment and their solvent accessibility impact the predictions made. Nevertheless, our analysis underscores the necessity of incorporating features related to non-LoFs in improving pathogenicity predictions. This can be achieved through scrutiny and inclusion of gene-level and protein-level features specific to each of the mutation mechanisms in question, as documented by <xref ref-type="bibr" rid="B55">Sevim Bayrak et al. (2021)</xref>. In addition, proteins in both AD and AR datasets reportedly show significant differences in functional class prevalence (<xref ref-type="bibr" rid="B15">Gerasimavicius et al., 2022</xref>), necessitating function-specific analysis to delineate characteristics of the disease mechanisms of mutations (<xref ref-type="bibr" rid="B20">Iqbal et al., 2020</xref>).</p>
<p>Our current method does not include explicit modeling of mutations into the protein structure, nor inclusion of protein dynamics, an inherent property linked to protein function. Indeed, inclusion of such details can aid in the recognition of the extent of mutation-induced changes in intra-protein structural contacts, as well as changes in thermodynamic stability (<xref ref-type="bibr" rid="B51">Rodrigues et al., 2018</xref>). In combination with other relevant features, these may provide considerable insights into understanding different effects across different mutation types, even with limited protein structural data. While we acknowledge the limitations of training our model on static protein microenvironments, we understand that more features may not necessarily imply better performance with neural networks. With suitable representations of protein structures (graphs) and information on protein dynamics it is important to address fundamental problems, such as predicting functional sites (<xref ref-type="bibr" rid="B9">Chiang et al., 2022</xref>) or predicting structurally important sites to further our understanding of model-driven approaches. This can help gauge utility of protein dynamics-informed or physics-informed graph representations in predicting variant pathogenicity.</p>
<p>To summarize, we have described a structure-guided approach to predict functional outcomes of missense variants using 3D-CNNs. We analyze and demonstrate the contribution of different features on the predictive ability of the neural network. Of particular note is the influence of evolutionary information of the variant neighborhood and their solvent accessibilities in determining variant pathogenicity. We further provide detailed assessment of our model&#x2019;s generalizability on distinct mechanisms of mutations, which presents a complex but critical challenge in improving pathogenicity predictions. Our analysis presents lessons to consider when using model-driven approaches to address questions in structure-guided predictions of variant pathogenicity.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s10">Supplementary Material</xref>, further inquiries can be dirrected to the corresponding author. The source code and documentation of DeepRank-Mut are available at <ext-link ext-link-type="uri" xlink:href="https://github.com/DeepRank/DeepRank-Mut/">https://github.com/DeepRank/DeepRank-Mut/</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>MH, LX, and BV conceived the project. CB and GR designed the algorithm from its parent. GR, CB, RvH, and JH implemented and evaluated the algorithm. BV and SH compiled and pruned datasets of missense variants and 3D structures. GR performed the analyses, interpretation of data and wrote the manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This research was supported by the Europees Fonds voor Regionale Ontwikkeling (EFRO) (R0005582). LX acknowledges support from Hypatia Fellowship from RadboudUMC (Rv819.52706). The work was carried out on the National Computer Facilities (NWO-2021.047).</p>
</sec>
<ack>
<p>The authors acknowledge Dario Marzella for his inputs on grid feature visualizations. The authors also acknowledge Dr. Peter-Bram t&#x2019;Hoen, Dr. Hanka Vensalaar, Daniel Rademaker and the reviewers for their useful suggestions.</p>
</ack>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>Authors SH and BV were employed by Bio-Prodict. Authors RvH and JH were employed by Vartion.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmolb.2023.1204157/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmolb.2023.1204157/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table2.xlsx" id="SM1" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table3.xlsx" id="SM2" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.xlsx" id="SM3" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet1.docx" id="SM4" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abrus&#xe1;n</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Marsh</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Alpha helices are more robust to mutations than beta strands</article-title>. <source>PLOS Comput. Biol.</source> <volume>12</volume>, <fpage>e1005242</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005242</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adzhubei</surname>
<given-names>I. A.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Peshkin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ramensky</surname>
<given-names>V. E.</given-names>
</name>
<name>
<surname>Gerasimova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bork</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>A method and server for predicting damaging missense mutations</article-title>. <source>Nat. Methods</source> <volume>7</volume>, <fpage>248</fpage>&#x2013;<lpage>249</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth0410-248</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adzhubei</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Jordan</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Sunyaev</surname>
<given-names>S. R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Predicting functional effect of human missense mutations using PolyPhen-2</article-title>. <source>Curr. Protoc. Hum. Genet.</source> <volume>7</volume>, <fpage>Unit7.20</fpage>. <pub-id pub-id-type="doi">10.1002/0471142905.hg0720s76</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altschul</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Madden</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Sch&#xe4;ffer</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>1997</year>). <article-title>Gapped BLAST and PSI-BLAST: A new generation of protein database search programs</article-title>. <source>Nucleic Acids Res.</source> <volume>25</volume>, <fpage>3389</fpage>&#x2013;<lpage>3402</lpage>. <pub-id pub-id-type="doi">10.1093/nar/25.17.3389</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ancien</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Pucci</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Godfroid</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rooman</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Prediction and interpretation of deleterious coding variants in terms of protein structural stability</article-title>. <source>Sci. Rep.</source> <volume>8</volume>, <fpage>4480</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-22531-2</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bagley</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Altman</surname>
<given-names>R. B.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Characterizing the microenvironment surrounding protein sites</article-title>. <source>Protein Sci.</source> <volume>4</volume>, <fpage>622</fpage>&#x2013;<lpage>635</lpage>. <pub-id pub-id-type="doi">10.1002/pro.5560040404</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Capriotti</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Altman</surname>
<given-names>R. B.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Improving the prediction of disease-related variants using protein three-dimensional structure</article-title>. <source>BMC Bioinforma.</source> <volume>12</volume>, <fpage>S3</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-12-S4-S3</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carter</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Douville</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Stenson</surname>
<given-names>P. D.</given-names>
</name>
<name>
<surname>Cooper</surname>
<given-names>D. N.</given-names>
</name>
<name>
<surname>Karchin</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Identifying mendelian disease genes with the variant effect scoring tool</article-title>. <source>BMC Genomics</source> <volume>14</volume>, <fpage>S3</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-14-S3-S3</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hui</surname>
<given-names>W.-H.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>S.-W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Encoding protein dynamic information in graph representation for functional residue identification</article-title>. <source>Cell Rep. Phys. Sci.</source> <volume>3</volume>, <fpage>100975</fpage>. <pub-id pub-id-type="doi">10.1016/j.xcrp.2022.100975</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chicco</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Jurman</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The advantages of the Matthews correlation coefficient (MCC) over F1 score and accuracy in binary classification evaluation</article-title>. <source>BMC Genomics</source> <volume>21</volume>, <fpage>6</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-019-6413-7</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>A. P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>PROVEAN web server: A tool to predict the functional effect of amino acid substitutions and indels</article-title>. <source>Bioinformatics</source> <volume>31</volume>, <fpage>2745</fpage>&#x2013;<lpage>2747</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv195</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Juan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Pazos</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Valencia</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Emerging methods in protein co-evolution</article-title>. <source>Nat. Rev. Genet.</source> <volume>14</volume>, <fpage>249</fpage>&#x2013;<lpage>261</lpage>. <pub-id pub-id-type="doi">10.1038/nrg3414</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feinauer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Weigt</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Context-aware prediction of pathogenicity of missense mutations involved in human disease</article-title>. <source>Arxiv</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1701.07246</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Findlay</surname>
<given-names>G. M.</given-names>
</name>
<name>
<surname>Daza</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Martin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Leith</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Gasperini</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Accurate classification of BRCA1 variants with saturation genome editing</article-title>. <source>Nature</source> <volume>562</volume>, <fpage>217</fpage>&#x2013;<lpage>222</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-018-0461-z</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gerasimavicius</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Livesey</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Marsh</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Loss-of-function, gain-of-function and dominant-negative mutations have profoundly different effects on protein structure</article-title>. <source>Nat. Commun.</source> <volume>13</volume>, <fpage>3895</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-022-31686-6</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gunning</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Fryer</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Fasham</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Crosby</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Ellard</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Baple</surname>
<given-names>E. L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Assessing performance of pathogenicity predictors using clinically relevant variant datasets</article-title>. <source>J. Med. Genet.</source> <volume>58</volume>, <fpage>547</fpage>&#x2013;<lpage>555</lpage>. <pub-id pub-id-type="doi">10.1136/jmedgenet-2020-107003</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heijl</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vroling</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>van den Bergh</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Joosten</surname>
<given-names>H.-J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Mind the gap: Preventing circularity in missense variant prediction</article-title>. <source>Biorxiv</source>. <pub-id pub-id-type="doi">10.1101/2020.05.06.080424</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hopf</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Sch&#xe4;rfe</surname>
<given-names>C. P. I.</given-names>
</name>
<name>
<surname>Rodrigues</surname>
<given-names>J. P. G. L. M.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Kohlbacher</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Sequence co-evolution gives 3D contacts and structures of protein complexes</article-title>. <source>Elife</source> <volume>3</volume>, <fpage>e03430</fpage>. <pub-id pub-id-type="doi">10.7554/eLife.03430</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hopf</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Ingraham</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Poelwijk</surname>
<given-names>F. J.</given-names>
</name>
<name>
<surname>Sch&#xe4;rfe</surname>
<given-names>C. P. I.</given-names>
</name>
<name>
<surname>Springer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Mutation effects predicted from sequence co-variation</article-title>. <source>Nat. Biotechnol.</source> <volume>35</volume>, <fpage>128</fpage>&#x2013;<lpage>135</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.3769</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iqbal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>P&#xe9;rez-Palma</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Jespersen</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>May</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Hoksza</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Heyne</surname>
<given-names>H. O.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Comprehensive characterization of amino acid positions in protein structures reveals molecular effect of missense variants</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>117</volume>, <fpage>28201</fpage>&#x2013;<lpage>28211</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.2002660117</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ittisoponpisan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Islam</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Khanna</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Alhuzimi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>David</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sternberg</surname>
<given-names>M. J. E.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Can predicted protein 3D structures provide reliable insights into whether missense variants are disease associated?</article-title> <source>J. Mol. Biol.</source> <volume>431</volume>, <fpage>2197</fpage>&#x2013;<lpage>2212</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2019.04.009</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jorgensen</surname>
<given-names>W. L.</given-names>
</name>
<name>
<surname>Tirado-Rives</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1988</year>). <article-title>The OPLS [optimized potentials for liquid simulations] potential functions for proteins, energy minimizations for crystals of cyclic peptides and crambin</article-title>. <source>J. Am. Chem. Soc.</source> <volume>110</volume>, <fpage>1657</fpage>&#x2013;<lpage>1666</lpage>. <pub-id pub-id-type="doi">10.1021/ja00214a001</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kabsch</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>1983</year>). <article-title>Dictionary of protein secondary structure: Pattern recognition of hydrogen-bonded and geometrical features</article-title>. <source>Biopolymers</source> <volume>22</volume>, <fpage>2577</fpage>&#x2013;<lpage>2637</lpage>. <pub-id pub-id-type="doi">10.1002/bip.360221211</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karczewski</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Francioli</surname>
<given-names>L. C.</given-names>
</name>
<name>
<surname>Tiao</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cummings</surname>
<given-names>B. B.</given-names>
</name>
<name>
<surname>Alf&#xf6;ldi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>The mutational constraint spectrum quantified from variation in 141,456 humans</article-title>. <source>Nature</source> <volume>581</volume>, <fpage>434</fpage>&#x2013;<lpage>443</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2308-7</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kircher</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Witten</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Jain</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>O&#x2019;Roak</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Cooper</surname>
<given-names>G. M.</given-names>
</name>
<name>
<surname>Shendure</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>A general framework for estimating the relative pathogenicity of human genetic variants</article-title>. <source>Nat. Genet.</source> <volume>46</volume>, <fpage>310</fpage>&#x2013;<lpage>315</lpage>. <pub-id pub-id-type="doi">10.1038/ng.2892</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Krizhevsky</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>ImageNet classification with deep convolutional neural networks</article-title>,&#x201d; in <source>Advances in neural information processing systems</source> (<publisher-name>Curran Associates, Inc</publisher-name>).</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kryukov</surname>
<given-names>G. V.</given-names>
</name>
<name>
<surname>Pennacchio</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Sunyaev</surname>
<given-names>S. R.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Most rare missense alleles are deleterious in humans: Implications for complex disease and association studies</article-title>. <source>Am. J. Hum. Genet.</source> <volume>80</volume>, <fpage>727</fpage>&#x2013;<lpage>739</lpage>. <pub-id pub-id-type="doi">10.1086/513473</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kucukkal</surname>
<given-names>T. G.</given-names>
</name>
<name>
<surname>Petukh</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Alexov</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Structural and physico-chemical effects of disease and non-disease nsSNPs on proteins</article-title>. <source>Curr. Opin. Struct. Biol.</source> <volume>32</volume>, <fpage>18</fpage>&#x2013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.1016/j.sbi.2015.01.003</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuipers</surname>
<given-names>R. K.</given-names>
</name>
<name>
<surname>Joosten</surname>
<given-names>H.-J.</given-names>
</name>
<name>
<surname>van Berkel</surname>
<given-names>W. J. H.</given-names>
</name>
<name>
<surname>Leferink</surname>
<given-names>N. G. H.</given-names>
</name>
<name>
<surname>Rooijen</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ittmann</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>3DM: Systematic analysis of heterogeneous superfamily data to discover protein functionalities</article-title>. <source>Proteins Struct. Funct. Bioinforma.</source> <volume>78</volume>, <fpage>2101</fpage>&#x2013;<lpage>2113</lpage>. <pub-id pub-id-type="doi">10.1002/prot.22725</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Landrum</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Benson</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Chao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chitipiralla</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>ClinVar: Improving access to variant interpretations and supporting evidence</article-title>. <source>Nucleic Acids Res.</source> <volume>46</volume>, <fpage>D1062</fpage>&#x2013;<lpage>D1067</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx1153</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laskowski</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Stephenson</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Sillitoe</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Orengo</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Thornton</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>VarSite: Disease variants and protein structure</article-title>. <source>Protein Sci.</source> <volume>29</volume>, <fpage>111</fpage>&#x2013;<lpage>119</lpage>. <pub-id pub-id-type="doi">10.1002/pro.3746</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lek</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Karczewski</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Minikel</surname>
<given-names>E. V.</given-names>
</name>
<name>
<surname>Samocha</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Banks</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Fennell</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> <collab>Exome Aggregation Consortium</collab> (<year>2016</year>). <article-title>Analysis of protein-coding genetic variation in 60,706 humans</article-title>. <source>Nature</source> <volume>536</volume>, <fpage>285</fpage>&#x2013;<lpage>291</lpage>. <pub-id pub-id-type="doi">10.1038/nature19057</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Krishnan</surname>
<given-names>V. G.</given-names>
</name>
<name>
<surname>Mort</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Xin</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kamati</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Cooper</surname>
<given-names>D. N.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Automated inference of molecular mechanisms of disease from amino acid substitutions</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>2744</fpage>&#x2013;<lpage>2750</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp528</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y. T.</given-names>
</name>
<name>
<surname>Capra</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Gerstein</surname>
<given-names>M. B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Predicting changes in protein thermodynamic stability upon point mutation with deep 3D convolutional neural networks</article-title>. <source>PLoS Comput. Biol.</source> <volume>16</volume>, <fpage>e1008291</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1008291</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jian</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Boerwinkle</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>dbNSFP: A lightweight database of human nonsynonymous SNPs and their functional predictions</article-title>. <source>Hum. Mutat.</source> <volume>32</volume>, <fpage>894</fpage>&#x2013;<lpage>899</lpage>. <pub-id pub-id-type="doi">10.1002/humu.21517</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>dbNSFP v4: a comprehensive database of transcript-specific functional predictions and annotations for human nonsynonymous and splice-site SNVs</article-title>. <source>Genome Med.</source> <volume>12</volume>, <fpage>103</fpage>. <pub-id pub-id-type="doi">10.1186/s13073-020-00803-9</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Livesey</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Marsh</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Using deep mutational scanning to benchmark variant effect predictors and identify disease mutations</article-title>. <source>Mol. Syst. Biol.</source> <volume>16</volume>, <fpage>e9380</fpage>. <pub-id pub-id-type="doi">10.15252/msb.20199380</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Livesey</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Marsh</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Interpreting protein variant effects with computational predictors and deep mutational scanning</article-title>. <source>Dis. Models Mech.</source> <volume>15</volume>, <fpage>dmm049510</fpage>. <pub-id pub-id-type="doi">10.1242/dmm.049510</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Loshchilov</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Hutter</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Decoupled weight decay regularization</article-title>. <source>Arxiv</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1711.05101</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marks</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Hopf</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Protein structure prediction from sequence variation</article-title>. <source>Nat. Biotechnol.</source> <volume>30</volume>, <fpage>1072</fpage>&#x2013;<lpage>1080</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.2419</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mitternacht</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>FreeSASA: An open source C library for solvent accessible surface area calculations</article-title>. <source>F1000Res</source> <volume>5</volume>, <fpage>189</fpage>. <pub-id pub-id-type="doi">10.12688/f1000research.7931.1</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morcos</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Pagnani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lunt</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bertolino</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Marks</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Direct-coupling analysis of residue coevolution captures native contacts across many protein families</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>108</volume>, <fpage>E1293</fpage>&#x2013;<lpage>E1301</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1111471108</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ng</surname>
<given-names>P. C.</given-names>
</name>
<name>
<surname>Henikoff</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>SIFT: Predicting amino acid changes that affect protein function</article-title>. <source>Nucleic Acids Res.</source> <volume>31</volume>, <fpage>3812</fpage>&#x2013;<lpage>3814</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkg509</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Niroula</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vihinen</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>How good are pathogenicity predictors in detecting benign variants?</article-title> <source>PLOS Comput. Biol.</source> <volume>15</volume>, <fpage>e1006481</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1006481</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pearson</surname>
<given-names>W. R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>An introduction to sequence similarity (&#x201c;Homology&#x201d;) searching</article-title>. <source>Curr. Protoc. Bioinforma. 0</source> <volume>3</volume>, <fpage>3.1.1</fpage>, <lpage>3.1.8</lpage>. <pub-id pub-id-type="doi">10.1002/0471250953.bi0301s42</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pincus</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Scheraga</surname>
<given-names>H. A.</given-names>
</name>
</person-group> (<year>1977</year>). <article-title>An approximate treatment of long-range interactions in proteins</article-title>. <source>J. Phys. Chem.</source> <volume>81</volume>, <fpage>1579</fpage>&#x2013;<lpage>1583</lpage>. <pub-id pub-id-type="doi">10.1021/j100531a013</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ponzoni</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pe&#xf1;aherrera</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Oltvai</surname>
<given-names>Z. N.</given-names>
</name>
<name>
<surname>Bahar</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Rhapsody: Predicting the pathogenicity of human missense variants</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>3084</fpage>&#x2013;<lpage>3092</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa127</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pun</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Ivanov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bellamy</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Montague</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>LaMont</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bradley</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Learning the shape of protein micro-environments with a holographic convolutional neural network</article-title>. <source>Arxiv</source>. <pub-id pub-id-type="doi">10.1101/2022.10.31.514614</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Renaud</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Geng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Georgievska</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ambrosetti</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ridder</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Marzella</surname>
<given-names>D. F.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>DeepRank: A deep learning framework for data mining 3D protein-protein interfaces</article-title>. <source>Nat. Commun.</source> <volume>12</volume>, <fpage>7068</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-021-27396-0</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reva</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Antipin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Predicting the functional impact of protein mutations: Application to cancer genomics</article-title>. <source>Nucleic Acids Res.</source> <volume>39</volume>, <fpage>e118</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkr407</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rodrigues</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Pires</surname>
<given-names>D. E.</given-names>
</name>
<name>
<surname>Ascher</surname>
<given-names>D. B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>DynaMut: Predicting the impact of mutations on protein conformation, flexibility and stability</article-title>. <source>Nucleic Acids Res.</source> <volume>46</volume>, <fpage>W350</fpage>&#x2013;<lpage>W355</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky300</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rost</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Conservation and prediction of solvent accessibility in protein families</article-title>. <source>Proteins</source> <volume>20</volume>, <fpage>216</fpage>&#x2013;<lpage>226</lpage>. <pub-id pub-id-type="doi">10.1002/prot.340200303</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Savojardo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Manfredi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Martelli</surname>
<given-names>P. L.</given-names>
</name>
<name>
<surname>Casadio</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Solvent accessibility of residues undergoing pathogenic variations in humans: From protein structures to protein sequences</article-title>. <source>Front. Mol. Biosci.</source> <volume>7</volume>, <fpage>626363</fpage>. <pub-id pub-id-type="doi">10.3389/fmolb.2020.626363</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schwarz</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Cooper</surname>
<given-names>D. N.</given-names>
</name>
<name>
<surname>Schuelke</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Seelow</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>MutationTaster2: Mutation prediction for the deep-sequencing age</article-title>. <source>Nat. Methods</source> <volume>11</volume>, <fpage>361</fpage>&#x2013;<lpage>362</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2890</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sevim Bayrak</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Stein</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Jain</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chaudhary</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Nadkarni</surname>
<given-names>G. N.</given-names>
</name>
<name>
<surname>Van Vleck</surname>
<given-names>T. T.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Identification of discriminative gene-level and protein-level features associated with pathogenic gain-of-function and loss-of-function variants</article-title>. <source>Am. J. Hum. Genet.</source> <volume>108</volume>, <fpage>2301</fpage>&#x2013;<lpage>2318</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2021.10.007</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shihab</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Gough</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cooper</surname>
<given-names>D. N.</given-names>
</name>
<name>
<surname>Stenson</surname>
<given-names>P. D.</given-names>
</name>
<name>
<surname>Barker</surname>
<given-names>G. L. A.</given-names>
</name>
<name>
<surname>Edwards</surname>
<given-names>K. J.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Predicting the functional, molecular, and phenotypic consequences of amino acid substitutions using hidden Markov models</article-title>. <source>Hum. Mutat.</source> <volume>34</volume>, <fpage>57</fpage>&#x2013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1002/humu.22225</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shorten</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Khoshgoftaar</surname>
<given-names>T. M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A survey on image data augmentation for deep learning</article-title>. <source>J. Big Data</source> <volume>6</volume>, <fpage>60</fpage>. <pub-id pub-id-type="doi">10.1186/s40537-019-0197-0</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shroff</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cole</surname>
<given-names>A. W.</given-names>
</name>
<name>
<surname>Diaz</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Morrow</surname>
<given-names>B. R.</given-names>
</name>
<name>
<surname>Donnell</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Annapareddy</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Discovery of novel gain-of-function mutations guided by structure-based deep learning</article-title>. <source>ACS Synth. Biol.</source> <volume>9</volume>, <fpage>2927</fpage>&#x2013;<lpage>2935</lpage>. <pub-id pub-id-type="doi">10.1021/acssynbio.0c00345</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Qu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Reetz</surname>
<given-names>M. T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Utility of B-factors in protein science: Interpreting rigidity, flexibility, and internal motion and engineering thermostability</article-title>. <source>Chem. Rev.</source> <volume>119</volume>, <fpage>1626</fpage>&#x2013;<lpage>1665</lpage>. <pub-id pub-id-type="doi">10.1021/acs.chemrev.8b00290</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thompson</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Spurdle</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Plazzer</surname>
<given-names>J.-P.</given-names>
</name>
<name>
<surname>Greenblatt</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Akagi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Al-Mulla</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Application of a 5-tiered scheme for standardized classification of 2,360 unique mismatch repair gene variants in the InSiGHT locus-specific database</article-title>. <source>Nat. Genet.</source> <volume>46</volume>, <fpage>107</fpage>&#x2013;<lpage>115</lpage>. <pub-id pub-id-type="doi">10.1038/ng.2854</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Torng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Altman</surname>
<given-names>R. B.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>3D deep convolutional neural networks for amino acid environment similarity analysis</article-title>. <source>BMC Bioinforma.</source> <volume>18</volume>, <fpage>302</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-017-1702-0</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Touw</surname>
<given-names>W. G.</given-names>
</name>
<name>
<surname>Baakman</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Black</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>te Beek</surname>
<given-names>T. A. H.</given-names>
</name>
<name>
<surname>Krieger</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Joosten</surname>
<given-names>R. P.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>A series of PDB-related databanks for everyday needs</article-title>. <source>Nucleic Acids Res.</source> <volume>43</volume>, <fpage>D364</fpage>&#x2013;<lpage>D368</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gku1028</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaser</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Adusumalli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Leng</surname>
<given-names>S. N.</given-names>
</name>
<name>
<surname>Sikic</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>P. C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>SIFT missense predictions for genomes</article-title>. <source>Nat. Protoc.</source> <volume>11</volume>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1038/nprot.2015.123</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veitia</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Caburet</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Birchler</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Mechanisms of mendelian dominance</article-title>. <source>Clin. Genet.</source> <volume>93</volume>, <fpage>419</fpage>&#x2013;<lpage>428</lpage>. <pub-id pub-id-type="doi">10.1111/cge.13107</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Venselaar</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Te Beek</surname>
<given-names>T. A. H.</given-names>
</name>
<name>
<surname>Kuipers</surname>
<given-names>R. K. P.</given-names>
</name>
<name>
<surname>Hekkelman</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Vriend</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Protein structure analysis of mutations causing inheritable diseases. An e-Science approach with life scientist friendly interfaces</article-title>. <source>BMC Bioinforma.</source> <volume>11</volume>, <fpage>548</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-11-548</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="book">
<collab>VKGL</collab> (<year>2019</year>). <source>Vereniging klinisch genetische laboratoriumdiagnostiek - home</source>. <comment>URL: <ext-link ext-link-type="uri" xlink:href="https://www.vkgl.nl/nl/">https://www.vkgl.nl/nl/</ext-link>
</comment> (<comment>accessed October 3, 2019)</comment>.</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vroling</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Heijl</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>White paper: The Helix pathogenicity prediction platform</article-title>. <source>Arxiv</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2104.01033</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Moult</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>SNPs, protein structure, and disease</article-title>. <source>Hum. Mutat.</source> <volume>17</volume>, <fpage>263</fpage>&#x2013;<lpage>270</lpage>. <pub-id pub-id-type="doi">10.1002/humu.22</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yates</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Filippis</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Kelley</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Sternberg</surname>
<given-names>M. J. E.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>SuSPect: Enhanced prediction of single amino acid variant (SAV) phenotype using network features</article-title>. <source>J. Mol. Biol.</source> <volume>426</volume>, <fpage>2692</fpage>&#x2013;<lpage>2701</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2014.04.026</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zardecki</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dutta</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Goodsell</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Lowe</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Voigt</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Burley</surname>
<given-names>S. K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>PDB-101: Educational resources supporting molecular explorations through biology and medicine</article-title>. <source>Protein Sci.</source> <volume>31</volume>, <fpage>129</fpage>&#x2013;<lpage>140</lpage>. <pub-id pub-id-type="doi">10.1002/pro.4200</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>