<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">887491</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.887491</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Inter-Residue Distance Prediction From Duet Deep Learning Models</article-title>
<alt-title alt-title-type="left-running-head">Zhang et al.</alt-title>
<alt-title alt-title-type="right-running-head">Deep-Learning-Based Residue Distance Prediction</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Huiling</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1538833/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Ying</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1702442/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bei</surname>
<given-names>Zhendong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ju</surname>
<given-names>Zhen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Meng</surname>
<given-names>Jintao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hao</surname>
<given-names>Min</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/444586/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Jingjing</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1557250/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Haiping</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/585952/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Xi</surname>
<given-names>Wenhui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/894491/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Shenzhen Institute of Advanced Technology</institution>, <institution>Chinese Academy of Sciences</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>University of Chinese Academy of Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>College of Electronic and Information Engineering</institution>, <institution>Southwest University</institution>, <addr-line>Chongqing</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/867328/overview">Ruiquan Ge</ext-link>, Hangzhou Dianzi University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1064518/overview">Leyi Wei</ext-link>, Shandong University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1714243/overview">Jun Wang</ext-link>, Nanjing University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/721368/overview">Duolin Wang</ext-link>, University of Missouri, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Wenhui Xi, <email>wh.xi@siat.ac.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Computational Genomics, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>887491</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>03</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Zhang, Huang, Bei, Ju, Meng, Hao, Zhang, Zhang and Xi.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Zhang, Huang, Bei, Ju, Meng, Hao, Zhang, Zhang and Xi</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Residue distance prediction from the sequence is critical for many biological applications such as protein structure reconstruction, protein&#x2013;protein interaction prediction, and protein design. However, prediction of fine-grained distances between residues with long sequence separations still remains challenging. In this study, we propose DuetDis, a method based on duet feature sets and deep residual network with squeeze-and-excitation (SE), for protein inter-residue distance prediction. DuetDis embraces the ability to learn and fuse features directly or indirectly extracted from the whole-genome/metagenomic databases and, therefore, minimize the information loss through ensembling models trained on different feature sets. We evaluate DuetDis and 11 widely used peer methods on a large-scale test set (610 proteins chains). The experimental results suggest that 1) prediction results from different feature sets show obvious differences; 2) ensembling different feature sets can improve the prediction performance; 3) high-quality multiple sequence alignment (MSA) used for both training and testing can greatly improve the prediction performance; and 4) DuetDis is more accurate than peer methods for the overall prediction, more reliable in terms of model prediction score, and more robust against shallow multiple sequence alignment (MSA).</p>
</abstract>
<kwd-group>
<kwd>residue distance prediction</kwd>
<kwd>protein structure reconstruction</kwd>
<kwd>deep learning</kwd>
<kwd>residual network</kwd>
<kwd>multiple sequence alignment</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Knowing the structure of a protein helps to understand the role of the protein, reveals how the protein performs its biological function, and also, sets the foundation for the protein&#x2019;s interaction with other molecules. Therefore, the knowledge of a protein&#x2019;s structure is very important for biology as well as for medicine and pharmacy. Since Anfinsen suggested that the advanced spatial structure of a protein is determined by its amino acid sequence (<xref ref-type="bibr" rid="B5">Anfinsen, 1973</xref>), it has been a &#x201c;holy grail&#x201d; for the computational biology community to develop an algorithm that can accurately predict a protein&#x2019;s structure from its amino acid sequence. Sequence-based residue contact/distance prediction plays a crucial role in protein structure reconstruction.</p>
<p>Residue&#x2013;residue contacts refer to the residue pairs that are close within a specific distance threshold in the three-dimensional protein structure. The contact map of a protein tells the constraints between residues in a binary form. Unlike the contact map, the distance map of a protein contains fine-grained information and, thus, provides more physical constraints of a protein structure. Protein contact/distance maps are 2D representations of the 3D protein structure and are being considered as one of the most important components in modern protein structure prediction packages. The application of predicted contacts/distances has been extended to intrinsic disorder region recognition (<xref ref-type="bibr" rid="B49">Schlessinger et al., 2007</xref>; <xref ref-type="bibr" rid="B52">Shimomura et al., 2019</xref>), protein&#x2013;protein interaction prediction (<xref ref-type="bibr" rid="B57">Vangone and Bonvin, 2015</xref>; <xref ref-type="bibr" rid="B14">Du et al., 2016</xref>; <xref ref-type="bibr" rid="B11">Cong et al., 2019</xref>), protein design (<xref ref-type="bibr" rid="B6">Anishchenko et al., 2021</xref>), etc.</p>
<p>Contact prediction methods in the early stage are mainly based on mutual information (MI) (<xref ref-type="bibr" rid="B42">Pollock and Taylor, 1997</xref>; <xref ref-type="bibr" rid="B15">Dunn et al., 2007</xref>; <xref ref-type="bibr" rid="B32">Lee and Kim, 2009</xref>), integer linear programming (ILP) techniques (<xref ref-type="bibr" rid="B38">McAllister and Floudas, 2008</xref>; <xref ref-type="bibr" rid="B44">Rajgaria et al., 2009</xref>; <xref ref-type="bibr" rid="B45">Rajgaria et al., 2010</xref>; <xref ref-type="bibr" rid="B61">Wei and Floudas, 2011</xref>), traditional machine learning (ML) algorithms (<xref ref-type="bibr" rid="B10">Cheng and Baldi, 2007</xref>; <xref ref-type="bibr" rid="B64">Wu and Zhang, 2008</xref>; <xref ref-type="bibr" rid="B56">Tegge et al., 2009</xref>), or techniques combining ILP with ML (<xref ref-type="bibr" rid="B60">Wang and Xu, 2013</xref>; <xref ref-type="bibr" rid="B71">Zhang et al., 2016</xref>). These methods are generally considered as local strategies since a residue pair is treated statistically independent of others (<xref ref-type="bibr" rid="B69">Zhang et al., 2020</xref>). Breakthroughs were achieved by capturing the correlated pattern of coevolved residues by global statistical inference methods such as direct coupling analysis (DCA) (<xref ref-type="bibr" rid="B62">Weigt et al., 2009</xref>) and sparse inverse covariance estimation (PSICOV) (<xref ref-type="bibr" rid="B25">Jones et al., 2012</xref>). Methods developed based on the ideas of DCA include EVfold (mfDCA) (<xref ref-type="bibr" rid="B41">Morcos et al., 2011</xref>), plmDCA (<xref ref-type="bibr" rid="B16">Ekeberg et al., 2013</xref>), GREMLIN (<xref ref-type="bibr" rid="B30">Kamisetty et al., 2013</xref>), CCMpred (<xref ref-type="bibr" rid="B50">Seemayer et al., 2014</xref>), gDCA (<xref ref-type="bibr" rid="B8">Baldassi et al., 2014</xref>), and Freecontact (<xref ref-type="bibr" rid="B29">Kaj&#xe1;n et al., 2014</xref>). These methods emphasize the importance of distinguishing between directly and indirectly correlated residues. Consensus-predictors like PconsC (<xref ref-type="bibr" rid="B54">Skwark et al., 2013</xref>), MetaPSICOV (<xref ref-type="bibr" rid="B27">Jones et al., 2014</xref>), and NeBcon (<xref ref-type="bibr" rid="B21">He et al., 2017</xref>) combine the output of different DCA-based or ML-based contact predictors to create consensus predictions. In recent years, the introduction of deep learning (DL) techniques has made tremendous progress for residue contact prediction. The DL-based contact map prediction algorithms are mainly based on convolutional neural networks (CNN) (such as DeepCov (<xref ref-type="bibr" rid="B26">Jones and Kandathil, 2018</xref>), DeepContact (<xref ref-type="bibr" rid="B36">Liu et al., 2018</xref>), and DNCON2 (<xref ref-type="bibr" rid="B3">Adhikari et al., 2018</xref>)), Unet [such as PconsC4 (<xref ref-type="bibr" rid="B40">Michel et al., 2019</xref>)], residual networks (ResNet) [such as DeepConPred2 (<xref ref-type="bibr" rid="B13">Ding et al., 2018</xref>), ResPRE (<xref ref-type="bibr" rid="B34">Li et al., 2019</xref>), MapPred (<xref ref-type="bibr" rid="B63">Wu et al., 2020</xref>) and TripletRes (<xref ref-type="bibr" rid="B35">Li et al., 2021</xref>)], ResNet combined with long short-term memory (LSTM) [such as SPOT-Contact (<xref ref-type="bibr" rid="B19">Hanson et al., 2018</xref>)] and transformers [such as ESM (<xref ref-type="bibr" rid="B46">Malinin and Gales, 2021</xref>) and SPOT-Contact-LM (<xref ref-type="bibr" rid="B53">Singh et al., 2022</xref>)]. COMTOP (<xref ref-type="bibr" rid="B48">Reza et al., 2021</xref>) uses the mixed ILP technique to combine different contact predictors (including several DL predictors) to further improve the prediction performance.</p>
<p>Although the predicted contacts have been successfully applied to the protein structure prediction packages (<xref ref-type="bibr" rid="B37">Marks et al., 2012</xref>; <xref ref-type="bibr" rid="B39">Michel et al., 2014</xref>; <xref ref-type="bibr" rid="B2">Adhikari et al., 2015</xref>; <xref ref-type="bibr" rid="B17">Gao et al., 2019</xref>), contact maps are still insufficient for accurate structure prediction. The reason is twofold. Most contact prediction methods use a cutoff of 8&#xa0;&#x212b; between C&#x3b2;-C&#x3b2; atoms to determine whether two residues are in contact or not, resulting a contact/non-contact ratio of less than 0.1 for globular proteins and a ratio of around 0.02 for alpha-helical transmembrane proteins (<xref ref-type="bibr" rid="B71">Zhang et al., 2016</xref>). The definition of contacts means that the native distance information is insufficiently being distinguished. Furthermore, contact-assisted conformation sampling may be misguided by several wrongly predicted contacts and needs a long time to generate good conformations for large proteins (<xref ref-type="bibr" rid="B66">Xu, 2019</xref>). In this context, inter-residue distance maps are more informative than residue&#x2013;residue contact maps since distances are fine-grained or real numbers, while contacts are binary values.</p>
<p>The methods for inter-residue distance prediction can be roughly categorized into two groups, those based on multiclass classification with discrete values and those based on regression with continuous values. Early distance maps are mainly predicted from homologous proteins (<xref ref-type="bibr" rid="B7">Asz&#xf3;di and Taylor, 1996</xref>) or from traditional machine learning techniques (<xref ref-type="bibr" rid="B58">Walsh et al., 2009</xref>; <xref ref-type="bibr" rid="B73">Zhao and Xu, 2012</xref>; <xref ref-type="bibr" rid="B31">Kukic et al., 2014</xref>). The introduction of deep learning technology has injected new life into distance prediction. <xref ref-type="bibr" rid="B59">Wang et al. (2017)</xref> pioneered the study of introducing residual network to multiclass distance prediction. The success of this approach can be partially attributed to the ability of deep learning to simultaneously consider the global set of pair-wise interactions instead of considering only one interaction at a time, thereby leading to more accurate discrimination between direct and indirect contacts. TripletRes (<xref ref-type="bibr" rid="B35">Li et al., 2021</xref>), which uses a similar deep learning architecture but with a unique set of features that include multiple coevolutionary coupling matrices directly deduced from deep multiple sequence alignment (MSA) without post-processing. GANProDist (<xref ref-type="bibr" rid="B12">Ding and Gong, 2020</xref>) predicts real value distance as a regression problem by generative adversarial network. PDNET (<xref ref-type="bibr" rid="B1">Adhikari, 2020</xref>), DeepDist (<xref ref-type="bibr" rid="B65">Wu et al., 2021</xref>), SDP (<xref ref-type="bibr" rid="B43">Rahman et al., 2022</xref>), and <xref ref-type="bibr" rid="B35">Li et al. (2021)</xref> (<xref ref-type="bibr" rid="B33">Li and Xu, 2021</xref>) predict both real-valued and binned distances from residual networks. DL-based distance prediction has recently demonstrated unprecedented ability to assist protein structure reconstruction such as DMPFold (<xref ref-type="bibr" rid="B18">Greener et al., 2019</xref>), RaptorX (<xref ref-type="bibr" rid="B66">Xu, 2019</xref>), trRosetta (<xref ref-type="bibr" rid="B67">Yang et al., 2020</xref>), and AlhpaFold (<xref ref-type="bibr" rid="B51">Senior et al., 2020</xref>). However, further progress needs more accurate inter-residue distance prediction since the quality of a predicted protein structure highly depends on the accuracy of the distance prediction.</p>
<p>
<xref ref-type="bibr" rid="B52">Shimomura et al. (2019)</xref> introduced a technique for predicting structurally disordered regions in proteins through average distance maps (AMD) based on statistics of average distances between residues. AMD first divides the residue pairs into different ranges according to their sequence separations, and calculates the distances of residue pairs within each range. AMD contact density maps were plotted against distance thresholds in different ranges. AMD technology detects the boundaries of structurally compact regions and finally predicts structurally disordered regions by calculating differences in density maps. The accuracy of AMD technology is comparable to the leading methods in the CASP competition such as PrDOS, DISOPRED, and Biomine. Protein domains are subunits that can fold and function independently. Therefore, correct domain boundary assignment is a critical step to achieve accurate protein structure and function analysis. <xref ref-type="bibr" rid="B74">Zheng et al. (2020)</xref> proposed FUPred to detect protein domains based on contact maps predicted by deep learning. The core idea of this method is to retrieve domain boundary locations by maximizing the number of intra-domain contacts while minimizing the number of inter-domain contacts from the contact map. FUpred was tested on a large-scale dataset consisting of 2,549 proteins and achieved a Matthews correlation coefficient (MCC) of 0.799 for single domain and multi-domain classification, which is 19.1% higher than the best machine learning-based method. For proteins with discontinuous domains, FUPred domain boundary detection and normalized domain overlap scores were 0.788 and 0.521, which were 17.3% and 23.8% higher than the best peer method. The results demonstrate that residue contact prediction provides a new way to accurately detect domains, especially discontinuous multi-domains. <xref ref-type="bibr" rid="B11">Cong et al. (2019)</xref> first compared the contact prediction methods based on mutual information, evolutionary coupling analysis, and deep learning in the prediction of residue contacts between protein complex chains and found that although the deep learning methods are outstanding for monomer contact prediction, they fail to outperform methods based on mutual information and evolutionary coupling analysis in inter-chain contact prediction. By identifying coevolving residue pairs between protein chains based on mutual information and evolutionary coupling analysis methods, 1,618 protein interactions (682 of which were unexpected) in <italic>Escherichia coli</italic>, and 911 protein interactions in <italic>M. tuberculosis</italic> (most of which were not identified in previous studies) were detected. The expected false positive rate for this study is between 10% and 20%, and the predicted interactions and networks provide a good starting point for further research. <xref ref-type="bibr" rid="B6">Anishchenko et al. (2021)</xref> investigated whether the residue distance information captured by deep neural networks is rich enough to generate new folded proteins. The study generated random amino acid sequences that were completely unrelated to the sequences of the native proteins used in the trRosetta training model, and fed them into the trRosetta structure prediction network to predict the starting residue distance map. Monte Carlo sampling is then performed in the amino acid sequence space to optimize the contrast between the network-predicted distribution of inter-residue distances and the background distribution averaged across all proteins. Optimization from different random starting points yields novel proteins spanning a broad range of sequences and predicted structures. Synthetic genes encoding 129 of the &#x2018;network-hallucinated&#x2019; sequences were obtained, and the proteins were expressed and purified in <italic>E. coli</italic>; 27 of the proteins yielded monodisperse species with circular dichroism spectra consistent with the hallucinated structures. Three of the three-dimensional structures of the hallucinated proteins were determined by experiments, and these closely matched the hallucinated models. We can see that residue distance-assisted protein structure prediction methods can be inverted to <italic>de novo</italic> protein design.</p>
<p>In this study, we develop a method based on deep residual convolutional neural network, named DuetDis, to predict the full-length multiclass distance map from a sequence. DuetDis uses a modified ResNet module to build the network, and adopts two sets of complementary feature sets to further improve the prediction accuracy. The results by DuetDis suggest that prediction results from different feature sets show obvious differences and ensembles of different feature sets can improve the prediction performance. DuetDis is also evaluated together with 11 widely used contact/distance prediction methods, and the results show that DuetDis is more accurate for the overall prediction, more reliable in terms of model prediction score, and more robust against shallow MSA. DuetDis is available at <ext-link ext-link-type="uri" xlink:href="http://hpcc.siat.ac.cn/hlzhang/DuetDis/">http://hpcc.siat.ac.cn/hlzhang/DuetDis/</ext-link>.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methods</title>
<sec id="s2-1">
<title>Datasets</title>
<p>The test set is obtained from our previous work, containing 610 highly non-redundant protein chains (<xref ref-type="bibr" rid="B68">Zhang et al., 2021</xref>). The training set is obtained through culling from the whole PDB with the following criteria: 1) with maximum sequence identity of 30% against each chain in the training set and test set; 3) with structure resolutions better than 2.5&#xa0;&#xc5;; 4) released before 1 May 2018 (before the beginning of CASP13). Finally, we get a non-redundant training set with 13,069 protein chains.</p>
</sec>
<sec id="s2-2">
<title>Definition of Contact and Distance</title>
<p>In this study, the definition of contacts is directly taken from the CASP experiments. A pair of residues in the experimental structure is considered to be in contact if the distance between their C&#x3b2; atoms (C&#x251; for Gly) is less than or equal to 8&#xa0;&#xc5;. For direct comparison, the multiclass distance definition is taken directly from trRosetta (<xref ref-type="bibr" rid="B1">Adhikari, 2020</xref>). The C&#x3b2;&#x2013;C&#x3b2; distance of every pair of residues in a target protein is treated as a vector of probabilities. The distance range (2&#x2013;20&#xa0;&#xc5;) is binned into 36 equally spaced segments, 0.5&#xa0;&#xc5; each, and one bin indicating that residues are not in contact, generating a distance vector of 37 bins for each residue pair.</p>
<p>Depending on the separation of two residues along the sequence (<italic>seq_sep</italic>), the contacts are classified into four classes: all-range (<italic>seq_sep</italic> &#x2265;6), short-range (6&#x2264; <italic>seq_sep</italic> &#x3c;12), medium-range (12&#x2264; <italic>seq_sep</italic> &#x3c;24), and long-range (<italic>seq_sep</italic> &#x3e;24).</p>
</sec>
<sec id="s2-3">
<title>Multiple Sequence Alignment Generation for Training and Test</title>
<p>Generating high-quality MSA is the first step for protein structure prediction based on the fact that interacting residue pairs are under evolutionary pressure to maintain the structure. The MSA used for model training is obtained as indicated in <xref ref-type="fig" rid="F1">Figure 1</xref>. The target sequence in the training set is searched against NCBI-nr (Jackhmmer), MetaClust (Jackhmmer), and BFD (HHblits) respectively, with <italic>E-</italic>values of 1e&#x2212;10 and 1e&#x2212;3. The search will stop if the target MSA has <italic>N</italic>
<sub>
<italic>seq</italic>
</sub>&#x3e; 25&#x2a;L (L is the sequence length) and <italic>N</italic>
<sub>
<italic>eff</italic>
</sub> &#x3e; 8&#x2a;L, where <italic>N</italic>
<sub>
<italic>seq</italic>
</sub> is the number of sequences (with sequence coverage &#x3e;50%) and <italic>N</italic>
<sub>
<italic>eff</italic>
</sub> [defined in (<xref ref-type="bibr" rid="B68">Zhang et al., 2021</xref>)] is the number of effective sequences in the MSA. After the search, the final MSA is obtained through sequence clustering (with sequence identity of 95%) using our in-house software nGIA (<xref ref-type="bibr" rid="B28">Ju et al., 2021</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The flowchart of MSA generation for the training set.</p>
</caption>
<graphic xlink:href="fgene-13-887491-g001.tif"/>
</fig>
<p>The MSA used for testing is obtained through searching JackHMMER (<xref ref-type="bibr" rid="B24">Johnson et al., 2010</xref>) against the NCBI-nr database with iteration &#x3d; 3 and <italic>E</italic>-value &#x3d; 0.0001.</p>
</sec>
<sec id="s2-4">
<title>Input Features</title>
<p>We used two subsets of features as the inputs for the deep residual network of DuetDis. The first feature set contains 526 feature channels: one-hot-encoder of the target sequence (1D features, 20&#x2a;2 channels); position-specific frequency matrix (1D features, 21&#x2a;2 channels, considering gap) and positional entropy (<xref ref-type="bibr" rid="B67">Yang et al., 2020</xref>) (1D features, 1&#x2a;2 channels); and coupling features (<xref ref-type="bibr" rid="B67">Yang et al., 2020</xref>) (2D features, 441 channels) derived from the inverse of the shrunk covariance matrix of MSA. The second feature set contains 151 feature channels: one-hot-encoder of the target sequence (1D features, 20&#x2a;2 channels), position-specific scoring matrix (<xref ref-type="bibr" rid="B4">Altschul et al., 1997</xref>) (1D features; 20&#x2a;2 channels; not considering gap), HMM profile (<xref ref-type="bibr" rid="B47">Remmert et al., 2012</xref>) (1D features, 30&#x2a;2 channels), secondary structure from SPOT-1D (Hanson et al., 2019) (1D features, 3&#x2a;2 channels), solvent accessible surface area from SPOT-1D (<xref ref-type="bibr" rid="B20">Hanson et al., 2019</xref>) (1D features, 1&#x2a;2 channels), CCMPRED score (Seemayer et al., 2014) (2D features, 1 channel), mutual information (<xref ref-type="bibr" rid="B70">Zhang et al., 2022</xref>) (2D feature, 1 channel), and statistical pair-wise contact potential (<xref ref-type="bibr" rid="B9">Betancourt and Thirumalai, 1999</xref>) (2D feature, 1 channel). The first feature set, indicated as FeatSet1, is mainly composed of 2D direct coupling features (441 out of 526 total features) from the MSA, while the second feature set, indicated as FeatSet2, is mainly composed of 1D sequence-based features (148 out of 151 total features). Most of the features except the one-hot-encoder features in FeatSet1 and FeatSet2 are different, so the prediction results from the two feature sets can be complementary in a duet way (as indicated in the results).</p>
<p>Both FeatSet1 and FeatSet2 are widely used by previous works (<xref ref-type="bibr" rid="B19">Hanson et al., 2018</xref>; <xref ref-type="bibr" rid="B67">Yang et al., 2020</xref>; <xref ref-type="bibr" rid="B23">Jain et al., 2021</xref>; <xref ref-type="bibr" rid="B55">Su et al., 2021</xref>), showing their great efficacy in contact/distance prediction. The aim of DuetDis is not to design new feature types, but to evaluate the performance of previously widely used feature sets under the situation of unified input and identical network, as well to study how to complement the advantages of different types of features for better prediction performance.</p>
</sec>
<sec id="s2-5">
<title>Deep Network Architectures and Model Training for Distance Prediction</title>
<p>The proposed method DuetDis implements residual neural networks (ResNet) (<xref ref-type="bibr" rid="B22">He et al., 2016</xref>) as the deep learning model. Compared to traditional convolutional networks, ResNet adds feedforward neural networks to an identity map of input, which helps enable the efficient training of extremely deep neural networks. ResNet has shown its power in successful residue contact/distance prediction (<xref ref-type="bibr" rid="B66">Xu, 2019</xref>; <xref ref-type="bibr" rid="B35">Li et al., 2021</xref>). The deep residual network of DuetDis is shown in <xref ref-type="fig" rid="F2">Figure 2A</xref>. The basic module of DuetDis network is a combination of squeeze-and-excitation and ResNet (SEResNet). The DuetDis network is composed of 33 SEResNet modules. In order to observe the impact of different networks and features on the prediction performance, we also designed another reference network (<xref ref-type="fig" rid="F2">Figure 2B</xref>), which has very different basic modules and backbones from <xref ref-type="fig" rid="F2">Figure 2A</xref>. The reference network is composed of 16 Res2Net modules. In this work, both SEResNet and Res2Net use dilation convolutions, while SEResNet use gelu and Res2Net use relu as the activation functions. The networks in <xref ref-type="fig" rid="F2">Figures 2A and B</xref> are indicated as Net1 and Net2, respectively. The final MSA obtained in <xref ref-type="fig" rid="F1">Figure 1</xref> is indicated as MSA_All, and a subset with top 10&#xa0;L sequences (ranked with sequence identity against the target sequence) selected from MSA_All is indicated as MSA_Top, and two disjoint subsets with each containing 10&#xa0;L sequences randomly selected from MSA_All are indicated as MSA_1 and MSA_2, respectively. As described in <xref ref-type="table" rid="T1">Table 1</xref>, 10 sub-models are trained based on Net1 (the DuetDis network) and Net2 (the reference network) with different feature sets from different MSAs. &#x201c;MSA Shuffle&#x201d; in <xref ref-type="table" rid="T1">Table 1</xref> means that the MSA are constructed through randomly selecting 10&#xa0;L sequences in MSA_All. For each epoch, N1_M1/N2_M1 are trained through &#x201c;MSA Shuffle&#x201d; strategy, N1_M2/N1_M3/N2_M2/N2_M3 are trained with MSA_Top, N1_M4/N2_M4 are trained with MSA_1, and N1_M5/N2_M5 are trained with MSA_2. The outputs of five sub-models are averaged to produce the final distance map, indicated as &#x201c;DuetAverage&#x201d; in <xref ref-type="fig" rid="F2">Figures 2A,B</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The network architecture used in this work. <bold>(A)</bold> The network used by DuetDis; <bold>(B)</bold> the reference network; <bold>(C)</bold> basic modules used in the networks; dilated convolution.</p>
</caption>
<graphic xlink:href="fgene-13-887491-g002.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The strategies used for the training of sub-models (N1_M1/N1_M2/N1_M3/N1_M4/N1_M5 are used for DuetDis).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Sub-models</th>
<th align="center">Network</th>
<th align="center">Feature set</th>
<th align="center">MSA</th>
<th align="center">MSA shuffle</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">N1_M1</td>
<td align="left">Net1</td>
<td align="left">FeatSet1</td>
<td align="left">MSA_All</td>
<td align="left">Yes</td>
</tr>
<tr>
<td align="left">N1_M2</td>
<td align="left">Net1</td>
<td align="left">FeatSet1</td>
<td align="left">MSA_Top</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">N1_M3</td>
<td align="left">Net1</td>
<td align="left">FeatSet2</td>
<td align="left">MSA_Top</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">N1_M4</td>
<td align="left">Net1</td>
<td align="left">FeatSet2</td>
<td align="left">MSA_1</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">N1_M5</td>
<td align="left">Net1</td>
<td align="left">FeatSet2</td>
<td align="left">MSA_2</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">N2_M1</td>
<td align="left">Net2</td>
<td align="left">FeatSet1</td>
<td align="left">MSA_All</td>
<td align="left">Yes</td>
</tr>
<tr>
<td align="left">N2_M2</td>
<td align="left">Net2</td>
<td align="left">FeatSet1</td>
<td align="left">MSA_Top</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">N2_M3</td>
<td align="left">Net2</td>
<td align="left">FeatSet2</td>
<td align="left">MSA_Top</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">N2_M4</td>
<td align="left">Net2</td>
<td align="left">FeatSet2</td>
<td align="left">MSA_1</td>
<td align="left">No</td>
</tr>
<tr>
<td align="left">N2_M5</td>
<td align="left">Net2</td>
<td align="left">FeatSet2</td>
<td align="left">MSA_2</td>
<td align="left">No</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The sub-models are generated by independent training branches. AdamW optimizer is performed with an initial learning rate of 0.0001 (multi-step decay is adopted as the learning rate decay strategy). Cross-entropy is used as the loss-function, and L2 regularization is used during the training process to correct overfitting. The training set is split into two parts: 600 protein chains are used as the validation set and the rest are used for training. The precision of top-L long-range contact predictions (multiclass distance map is converted to the binary contact map according to the definition in <xref ref-type="sec" rid="s2-2">Section 2.2</xref>) on the validation dataset is calculated at each epoch, and the training process will stop when there is no update of the validation precision for 10 epochs. The training processes are implemented in Pytorch on TeslaV100 SMX2, and each independent training generally takes 5&#x2013;10&#xa0;days.</p>
</sec>
<sec id="s2-6">
<title>Evaluation Metrics</title>
<p>
<list list-type="simple">
<list-item>
<p>1) The predicted distance map is a matrix of probability estimates. We analyze the performance of predictors on reduced lists of distances/contacts (sorted by the probability estimates) selected by either the probability threshold or the top-L/<italic>n</italic> (L is the sequence length, and <italic>n &#x3d;</italic> 1, 2, 5) criteria. The prediction performance is assessed using precision (accuracy in some references), coverage (recall in some references), and Matthew&#x2019;s Correlation Coefficient (MCC), defined as follows:</p>
</list-item>
</list>
<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <italic>TP</italic>, <italic>FP</italic>, <italic>TN</italic>, and <italic>FN</italic> are the number of true positive, false positive, true negative, and false negative contacts, respectively.<list list-type="simple">
<list-item>
<p>2) Standard deviation reflects the degree of dispersion among individuals within the group, which is defined as</p>
</list-item>
</list>
<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m5">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is the mean of the variable <italic>x</italic>. The standard deviation can be used to evaluate the dispersion of <italic>Precision</italic>, <italic>Coverage</italic>, and <italic>MCC.</italic>
<list list-type="simple">
<list-item>
<p>3) Jaccard index (Jaccard similarity coefficient) measures the similarities between sets. It is defined as the size of the intersection divided by the size of the union of two sets.</p>
</list-item>
</list>
<disp-formula id="e5">
<mml:math id="m6">
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x222a;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <italic>X</italic> and <italic>Y</italic> are the set of predicted contacts from two different predictors, <inline-formula id="inf2">
<mml:math id="m7">
<mml:mrow>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the number of elements in the intersection of <italic>X</italic> and <italic>Y</italic> and the <inline-formula id="inf3">
<mml:math id="m8">
<mml:mrow>
<mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x222a;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of elements in the union of <italic>X</italic> and <italic>Y</italic>. The Jaccard index has values in the range of [0,1], with the value of 0 for completely dissimilar ones and 1 for identical predictors.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<p>In this section, we assess the performance of DuetDis from different perspectives. <xref ref-type="sec" rid="s3-1">Section 3.1</xref>, <xref ref-type="sec" rid="s3-2">3.2</xref> study the performance of sub-models, while <xref ref-type="sec" rid="s3-3">Section 3.3</xref>&#x2013;<xref ref-type="sec" rid="s3-5">3.5</xref> focus on the comparison between DuetDis and peer methods. The peer methods used in this work are 4 DCA-based contact predictors (EVfold, FreeContact, gDCA, and CCMpred), 4 DL-based contact predictors (DeepCov, PconsC4, DNCON2, and SPOT-Contact), and 3 DL-based distance predictors (TripletRes, trRosetta, and RaptorX). <xref ref-type="sec" rid="s3-1">Section 3.1</xref>&#x2013;<xref ref-type="sec" rid="s3-3">3.3</xref> and <xref ref-type="sec" rid="s3-5">Section 3.5</xref> use the results of top-L/<italic>n</italic> (<italic>n</italic> &#x3d; 1, 2, 5) predictions, while <xref ref-type="sec" rid="s3-4">Section 3.4</xref> considers the results given by specific probability/score threshold. All sub-models and peer-methods use the same MSA as input.</p>
<sec id="s3-1">
<title>Prediction Results From Different Feature Sets Show Obvious Differences</title>
<p>We use the Jaccard indices of prediction results from 10 sub-models (as described in <xref ref-type="table" rid="T1">Table 1</xref>) to study their prediction similarities. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the dendrogram heatmap of Jaccard indices using Ward&#x2019;s hierarchical clustering method on the independent test set. The Jaccard index between two methods is calculated by averaging the Jaccard index value of each protein on the whole test set. According to the clustering results, these 10 sub-models can be roughly divided into two categories, and each category contains two sub-categories. N1_M1/ N1_M2 and N2_M1/ N2_M2 trained by FeatSet1 are clustered into one category (Category_1), while N1_M3/ N1_M4/ N1_M5 and N2_M3/ N2_M4/ N2_M5 trained by FeatSet2 form another category (Category_2). N1_M1/ N1_M2 trained by Net1 and N2_M1/ N2_M1 trained by Net2 form two sub-categories in Category_1, while N1_M3/ N1_M4/ N1_M5 trained by Net1 and N2_M3/ N2_M4/ N2_M5 trained by Net2 form two sub-categories in Category_2. So, we can draw the conclusion that prediction results from different feature sets show obvious differences, and the conclusion is true for all-range, short-range, mid-range, and long-range contacts/distances. The feature set decides the similarity between models for typical architectures of networks.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Prediction similarities between different sub-models for <bold>(A)</bold> all-range, <bold>(B)</bold> short-range, <bold>(C)</bold> mid-range, and <bold>(D)</bold> long-range contacts/distances.</p>
</caption>
<graphic xlink:href="fgene-13-887491-g003.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>Ensembling Different Feature Sets Improves Prediction Performance</title>
<p>The prediction accuracies of N1_M1/ N1_M2/ N1_M3/ N1_M4/ N1_M5/ N1_Ensemble (obtained by averaging the five Net1 sub-models) and N2_M1/ N2_M2/ N2_M3/ N2_M4/ N2_M5/ N2_Ensemble (obtained by averaging the five Net2 sub-models) are listed in <xref ref-type="table" rid="T2">Tables 2</xref>, <xref ref-type="table" rid="T3">3</xref>, respectively.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>The prediction precisions of N1_M1/N1_M2/N1_M3/N1_M4/N1_M5/N1_Ensemble for different sequence separations.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Range</th>
<th align="center">Method</th>
<th align="center">Top-L</th>
<th align="center">Top-L/2</th>
<th align="center">Top-L/5</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="6" align="left">All</td>
<td align="left">N1_M1</td>
<td align="char" char=".">0.7769</td>
<td align="char" char=".">0.8717</td>
<td align="char" char=".">0.9206</td>
</tr>
<tr>
<td align="left">N1_M2</td>
<td align="char" char=".">0.7587</td>
<td align="char" char=".">0.8475</td>
<td align="char" char=".">0.8941</td>
</tr>
<tr>
<td align="left">N1_M3</td>
<td align="char" char=".">0.7491</td>
<td align="char" char=".">0.846</td>
<td align="char" char=".">0.9027</td>
</tr>
<tr>
<td align="left">N1_M4</td>
<td align="char" char=".">0.7256</td>
<td align="char" char=".">0.8266</td>
<td align="char" char=".">0.8888</td>
</tr>
<tr>
<td align="left">N1_M5</td>
<td align="char" char=".">0.7319</td>
<td align="char" char=".">0.8328</td>
<td align="char" char=".">0.8942</td>
</tr>
<tr>
<td align="left">N1_Ensemble</td>
<td align="char" char=".">0.7896</td>
<td align="char" char=".">0.8786</td>
<td align="char" char=".">0.9266</td>
</tr>
<tr>
<td rowspan="6" align="left">Short</td>
<td align="left">N1_M1</td>
<td align="char" char=".">0.2955</td>
<td align="char" char=".">0.481</td>
<td align="char" char=".">0.7389</td>
</tr>
<tr>
<td align="left">N1_M2</td>
<td align="char" char=".">0.2928</td>
<td align="char" char=".">0.4754</td>
<td align="char" char=".">0.7287</td>
</tr>
<tr>
<td align="left">N1_M3</td>
<td align="char" char=".">0.2948</td>
<td align="char" char=".">0.4757</td>
<td align="char" char=".">0.7374</td>
</tr>
<tr>
<td align="left">N1_M4</td>
<td align="char" char=".">0.2824</td>
<td align="char" char=".">0.4588</td>
<td align="char" char=".">0.7109</td>
</tr>
<tr>
<td align="left">N1_M5</td>
<td align="char" char=".">0.2947</td>
<td align="char" char=".">0.473</td>
<td align="char" char=".">0.7219</td>
</tr>
<tr>
<td align="left">N1_Ensemble</td>
<td align="char" char=".">0.2988</td>
<td align="char" char=".">0.4918</td>
<td align="char" char=".">0.7633</td>
</tr>
<tr>
<td rowspan="6" align="left">Medium</td>
<td align="left">N1_M1</td>
<td align="char" char=".">0.3512</td>
<td align="char" char=".">0.5477</td>
<td align="char" char=".">0.7725</td>
</tr>
<tr>
<td align="left">N1_M2</td>
<td align="char" char=".">0.3422</td>
<td align="char" char=".">0.5336</td>
<td align="char" char=".">0.7514</td>
</tr>
<tr>
<td align="left">N1_M3</td>
<td align="char" char=".">0.342</td>
<td align="char" char=".">0.5329</td>
<td align="char" char=".">0.7533</td>
</tr>
<tr>
<td align="left">N1_M4</td>
<td align="char" char=".">0.3306</td>
<td align="char" char=".">0.5135</td>
<td align="char" char=".">0.7275</td>
</tr>
<tr>
<td align="left">N1_M5</td>
<td align="char" char=".">0.3371</td>
<td align="char" char=".">0.5209</td>
<td align="char" char=".">0.7352</td>
</tr>
<tr>
<td align="left">N1_Ensemble</td>
<td align="char" char=".">0.3537</td>
<td align="char" char=".">0.5592</td>
<td align="char" char=".">0.7895</td>
</tr>
<tr>
<td rowspan="6" align="left">Long</td>
<td align="left">N1_M1</td>
<td align="char" char=".">0.6245</td>
<td align="char" char=".">0.7696</td>
<td align="char" char=".">0.865</td>
</tr>
<tr>
<td align="left">N1_M2</td>
<td align="char" char=".">0.6062</td>
<td align="char" char=".">0.7411</td>
<td align="char" char=".">0.8273</td>
</tr>
<tr>
<td align="left">N1_M3</td>
<td align="char" char=".">0.594</td>
<td align="char" char=".">0.7308</td>
<td align="char" char=".">0.8246</td>
</tr>
<tr>
<td align="left">N1_M4</td>
<td align="char" char=".">0.5695</td>
<td align="char" char=".">0.7091</td>
<td align="char" char=".">0.8088</td>
</tr>
<tr>
<td align="left">N1_M5</td>
<td align="char" char=".">0.5742</td>
<td align="char" char=".">0.712</td>
<td align="char" char=".">0.8121</td>
</tr>
<tr>
<td align="left">N1_Ensemble</td>
<td align="char" char=".">0.6416</td>
<td align="char" char=".">0.7797</td>
<td align="char" char=".">0.8626</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The prediction precisions of N2_M1/N2_M2/N2_M3/N2_M4/N2_M5/N2_Ensemble for different sequence separations. -80</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Range</th>
<th align="center">Method</th>
<th align="center">Top-L</th>
<th align="center">Top-L/2</th>
<th align="center">Top-L/5</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="6" align="left">All</td>
<td align="left">N2_M1</td>
<td align="char" char=".">0.7532</td>
<td align="char" char=".">0.8562</td>
<td align="char" char=".">0.9103</td>
</tr>
<tr>
<td align="left">N2_M2</td>
<td align="char" char=".">0.7435</td>
<td align="char" char=".">0.839</td>
<td align="char" char=".">0.8938</td>
</tr>
<tr>
<td align="left">N2_M3</td>
<td align="char" char=".">0.7148</td>
<td align="char" char=".">0.8188</td>
<td align="char" char=".">0.8828</td>
</tr>
<tr>
<td align="left">N2_M4</td>
<td align="char" char=".">0.7091</td>
<td align="char" char=".">0.8119</td>
<td align="char" char=".">0.8768</td>
</tr>
<tr>
<td align="left">N2_M5</td>
<td align="char" char=".">0.7071</td>
<td align="char" char=".">0.8121</td>
<td align="char" char=".">0.879</td>
</tr>
<tr>
<td align="left">N2_Ensemble</td>
<td align="char" char=".">0.7590</td>
<td align="char" char=".">0.8579</td>
<td align="char" char=".">0.9153</td>
</tr>
<tr>
<td rowspan="6" align="left">Short</td>
<td align="left">N2_M1</td>
<td align="char" char=".">0.2864</td>
<td align="char" char=".">0.4654</td>
<td align="char" char=".">0.7172</td>
</tr>
<tr>
<td align="left">N2_M2</td>
<td align="char" char=".">0.2901</td>
<td align="char" char=".">0.4647</td>
<td align="char" char=".">0.71</td>
</tr>
<tr>
<td align="left">N2_M3</td>
<td align="char" char=".">0.2852</td>
<td align="char" char=".">0.4583</td>
<td align="char" char=".">0.7014</td>
</tr>
<tr>
<td align="left">N2_M4</td>
<td align="char" char=".">0.2831</td>
<td align="char" char=".">0.4547</td>
<td align="char" char=".">0.6982</td>
</tr>
<tr>
<td align="left">N2_M5</td>
<td align="char" char=".">0.2825</td>
<td align="char" char=".">0.4548</td>
<td align="char" char=".">0.7002</td>
</tr>
<tr>
<td align="left">N2_Ensemble</td>
<td align="char" char=".">0.3449</td>
<td align="char" char=".">0.5396</td>
<td align="char" char=".">0.7367</td>
</tr>
<tr>
<td rowspan="6" align="left">Medium</td>
<td align="left">N2_M1</td>
<td align="char" char=".">0.3413</td>
<td align="char" char=".">0.5325</td>
<td align="char" char=".">0.755</td>
</tr>
<tr>
<td align="left">N2_M2</td>
<td align="char" char=".">0.3428</td>
<td align="char" char=".">0.5267</td>
<td align="char" char=".">0.7395</td>
</tr>
<tr>
<td align="left">N2_M3</td>
<td align="char" char=".">0.3298</td>
<td align="char" char=".">0.5082</td>
<td align="char" char=".">0.7206</td>
</tr>
<tr>
<td align="left">N2_M4</td>
<td align="char" char=".">0.3281</td>
<td align="char" char=".">0.5042</td>
<td align="char" char=".">0.7152</td>
</tr>
<tr>
<td align="left">N2_M5</td>
<td align="char" char=".">0.3283</td>
<td align="char" char=".">0.5057</td>
<td align="char" char=".">0.7159</td>
</tr>
<tr>
<td align="left">N2_Ensemble</td>
<td align="char" char=".">0.3449</td>
<td align="char" char=".">0.5396</td>
<td align="char" char=".">0.7602</td>
</tr>
<tr>
<td rowspan="6" align="left">Long</td>
<td align="left">N2_M1</td>
<td align="char" char=".">0.6035</td>
<td align="char" char=".">0.746</td>
<td align="char" char=".">0.8473</td>
</tr>
<tr>
<td align="left">N2_M2</td>
<td align="char" char=".">0.5997</td>
<td align="char" char=".">0.7361</td>
<td align="char" char=".">0.828</td>
</tr>
<tr>
<td align="left">N2_M3</td>
<td align="char" char=".">0.5638</td>
<td align="char" char=".">0.7022</td>
<td align="char" char=".">0.8066</td>
</tr>
<tr>
<td align="left">N2_M4</td>
<td align="char" char=".">0.5525</td>
<td align="char" char=".">0.6877</td>
<td align="char" char=".">0.7913</td>
</tr>
<tr>
<td align="left">N2_M5</td>
<td align="char" char=".">0.5548</td>
<td align="char" char=".">0.6917</td>
<td align="char" char=".">0.7941</td>
</tr>
<tr>
<td align="left">N2_Ensemble</td>
<td align="char" char=".">0.6136</td>
<td align="char" char=".">0.7508</td>
<td align="char" char=".">0.8473</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As we can see from <xref ref-type="table" rid="T2">Table 2</xref>, N1_M1 trained through randomly shuffling MSA_All can obtain the best performance, which is 1.8%/ 0.3%/ 0.9%/ 1.8%, 2.8%/ 0.1%/ 0.9%/ 3.0%, 5.1%/ 1%/ 2.1%/ 5.5%, and 4.5%/ 0.1%/ 2.1%/ 5% higher than N2_M2/ N2_M3/ N2_M4/ N2_M5 for top-L all-/ short-/ medium-/ long-range predictions. Although using the same network and feature set, N1_M1 shows superior prediction precisions than N1_M2, implying that randomly shuffling MSA_All in each epoch enables augmentation of the training set and thus, a better model can be obtained. N1_M3 uses the same network and feature set as N1_M4 and N1_M5, but the prediction precisions of N1_M3 are higher than N1_M4 and N1_M5, indicating that high-quality MSA used for training helps to boost the model performance. N1_Ensemble outperforms the individual sub-models N1_M1/ N1_M2/ N1_M3/ N1_M4/ N1_M5 by 1.3%/ 3.1%/ 4.0%/ 6.4%/ 5.8%, 0.3%/ 0.6%/ 0.4%/ 1.6%/ 0.4%, 0.3%/ 1.2%/ 1.2%/ 2.3%/ 1.7%, and 1.7%/ 3.5%/ 4.8%/ 7.2%/ 6.7% for top-L all-/ short-/ medium-/ long-range predictions, suggesting that ensembles of models trained on different feature sets can improve the overall prediction performance. Similar phenomenon can be observed and consistent conclusions can be drawn from the results in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
</sec>
<sec id="s3-3">
<title>The Overall Performance of DuetDis</title>
<p>The prediction precisions of all-/ short-/ medium-/ long-range contacts for DuetDis and other 11 peer methods on the independent test set are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. In general, DL methods, which can capture the higher-order residue correlations and use nonlinear models with fewer parameters to be estimated from thousands of protein families (<xref ref-type="bibr" rid="B45">Rajgaria et al., 2010</xref>), significantly outperform DCA methods. Specifically, DuetDis shows the best overall performance. Compared with DeepCov/ PconsC4/ DNCON2/ SPOT/ TripletRes/ trRosetta/ RaptorX, DuetDis obtains 22.1%/ 18.8%/ 17.2%/ 3.5%/ 6.3%/ 3.8%/ 2.4%, 5.2%/ 5.2%/ 3.9%/ 1.0%/ 1.4%/ 0.8%/ 2.2%, 8.7%/ 7.5%/ 6.2%/ 1.4%/ 1.8%/ 1.4%/ 1.4%, and 2.4%/ 1.9%/ 3.8%/ 6.2%/ 3.8%/ 1.8% higher precisions for all-range, short-range, medium-range, and long-range top-L predictions, as well as 12.5%/ 13.3%/ 9.5%/ 1.9%/ 4.2%/ 2.7%/ 1.9%, and 17.6%/ 16.3%/ 13.1%/ 2.6%/ 6.6%/ 4.3%/ 3.4% higher precisions for all-range, short-range, medium-range, and long-range top-L/5 predictions, respectively. The better performance of DuetDis is probably due to the high-quality MSAs used for training, the delicately designed deep residual network, and the effective integration of different features.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The overall prediction precisions for <bold>(A)</bold> all-range, <bold>(B)</bold> short-range, <bold>(C)</bold> medium-range, and <bold>(D)</bold> long-range contacts/distances.</p>
</caption>
<graphic xlink:href="fgene-13-887491-g004.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>DuetDis Embraces High Model Reliability in Terms of Prediction Score</title>
<p>The confidence of the probability (score) given by a DCA or DL model can greatly reflect the reliability of the corresponding model. The prediction probabilities (scores) given by EVfold, FreeContact, gDCA, CCMpred, DeepCov, PconsC4, DNCON2, SPOT, TripletRes, trRosetta, RaptorX, and DuetDis are distributed at (0.000,1.309), (&#x2212;2.537,17.931), (&#x2212;1.243, 6.564), (0.000, 5.270), (0.0, 1.0), (0.0, 1.0), (0.0, 1.0), (0.0, 1.0), (0.0, 1.0), (0.0, 1.0), (0.0, 1.0), and (0.0, 1.0), respectively. For machine learning (both traditional and deep learning) applications, people usually use 0.5 as a threshold for classification. However, the threshold may be inaccurate for a complex problem like contact/distance prediction. Therefore, studying the scoring trend and the reliability of the model is of great benefit to understand the model performance.</p>
<p>
<xref ref-type="fig" rid="F5">Figure 5</xref> illustrates the prediction performance in terms of precision/ coverage/ MCC with the increase in probability (score) threshold given by DuetDis and the peer methods. With the increase of the probability (score) threshold, the prediction coverages decrease monotonically for all methods. As the threshold increases, their precision curves go down at some probability (score) value. The prediction precisions of all DL methods (DeepCov/ PconsC4/ DNCON2/ SPOT/ TripletRes/ trRosetta/ RaptorX) increase monotonically with the probability (score) threshold. However, the precision curves of DCA methods (EVfold/ FreeContact/ gDCA/ CCMpred) show turning points at some probability (score) values. Meanwhile, DCA methods also show much larger STDs on precisions and relatively lower coverages/MCCs compared with DL methods. The numbers under the precision curve in <xref ref-type="fig" rid="F4">Figure 4</xref> are the numbers of proteins with predictions returned using the corresponding probability (score) threshold on the x-axis. It is obvious that, as the probability (score) threshold increases, there are more proteins being predicted by DL methods than by DCA methods. Specifically, DuetDis achieves prediction precisions/ coverages/ MCCs of 98.1%/ 15.0%/ 0.352 (calculated on the 523 proteins with prediction scores higher than 0.95) at the (score) threshold of 0.95, which are higher than that by DeepCov (94.7%/ 7.4%/ 0.240: 431 proteins), PconsC4 (96.3%/ 6.5%/ 0.228: 448 proteins), DNCON2 (96.8%/ 4.4%/ 0.173: 396 proteins), SPOT (97.5%/ 12.3%/ 0.318: 544 proteins), TripletRes (93.0%/ 19.6%/ 0.399: 557 proteins), trRosetta (96.5%/ 9.4%/ 0.276: 513 proteins), and RaptorX (97.2%/ 14.5%/ 0.352: 497 proteins). In summary, DuetDis shows higher reliability in model probability (score) compared with peer methods.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Prediction performance in terms of precision, coverage, MCC, and the corresponding standard deviation (the shaded area around the curves) with the increasing probability (score) threshold given by the predictors. The numbers under the precision curve (blue) are the numbers of proteins with predictions returned using the corresponding (score) threshold on the x-axis.</p>
</caption>
<graphic xlink:href="fgene-13-887491-g005.tif"/>
</fig>
</sec>
<sec id="s3-5">
<title>DuetDis Is Robust Against Shallow Multiple Sequence Alignment</title>
<p>Coevolutionary coupling signals extracted from MSA play central role in most modern contact/distance prediction methods. In this study, the independent test set is divided into six groups according to <italic>N</italic>
<sub>
<italic>eff</italic>
</sub> (&#x3c;5, 5&#x2013;0.2&#xa0;L, 0.2&#xa0;L&#x2013;L, L&#x2013;5&#xa0;L, 5&#x2013;8&#xa0;L, and &#x3e;8&#xa0;L). The performance of different methods on these sub-groups of the test set is shown in <xref ref-type="fig" rid="F6">Figure 6</xref>. DuetDis achieves prediction precisions of 64.4% for <italic>N</italic>
<sub>
<italic>eff</italic>
</sub> &#x3c;5, 85.1% for <italic>N</italic>
<sub>
<italic>ef</italic>
</sub> &#x3d; 5&#x2013;0.2&#xa0;L (2.5% higher than the second), 92.5% for <italic>N</italic>
<sub>
<italic>eff</italic>
</sub> &#x3d; 0.2&#xa0;L&#x2013;L (0.5% higher than the second), 97.5% for <italic>N</italic>
<sub>
<italic>eff</italic>
</sub> &#x3d; L&#x2013;5&#xa0;L (0.8% higher than the second), 96.9% for <italic>N</italic>
<sub>
<italic>eff</italic>
</sub> &#x3d; 5&#x2013;8&#xa0;L (0.2% higher than the second), and 95.6% for <italic>N</italic>
<sub>
<italic>eff</italic>
</sub> &#x3d; 5&#x2013;8&#xa0;L (0.9% higher than the second). For <italic>N</italic>
<sub>
<italic>eff</italic>
</sub> &#x3c;5&#xa0;L, DuetDis ranks the second in prediction precision; while for <italic>N</italic>
<sub>
<italic>eff</italic>
</sub> &#x3d; 5&#x2013;0.2&#xa0;L, 0.2&#xa0;L&#x2013;L, L&#x2013;5&#xa0;L, 5&#x2013;8&#xa0;L and &#x3e;8&#xa0;L, DuetDis is in the leading position of prediction precision. For Neff &#x3c;5&#xa0;L, PconsC4 shows a STD of 0.125 which is smaller than DuetDis, however, the smaller STD is because of lower overall precision by PconsC4 (the average prediction precisions are 8.7% for PconsC4 and 64.4% for DuetDis). Hence, DuetDis obtains the least STD among all DL methods for all sub-groups of the test set. In general, DuetDis shows leading precisions and the smallest STD for most ranges of <italic>N</italic>
<sub>
<italic>eff</italic>
</sub>, especially highlights its robustness in shallow MSA-based distance prediction.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Prediction precisions of different methods for all-range, top-L, and top-L/5 predictions with the variation of <italic>N</italic>
<sub>
<italic>eff</italic>
</sub>. The error bar is the standard deviation of all precisions (for top L/5 predictions) in each sub-test set.</p>
</caption>
<graphic xlink:href="fgene-13-887491-g006.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>Conclusion</title>
<p>Proteins are considered as the molecular machines and perform many important functions of life (<xref ref-type="bibr" rid="B72">Zhang et al., 2017</xref>). Knowing the structure of a protein helps to understand the role of the protein, how the protein performs its biological function, and the interaction between the protein and the protein (or other molecules), which is very important for biology as well as for medicine and pharmacy. Residue distance prediction from the sequence is critical for many biological applications such as protein structure reconstruction. However, prediction of large distances and distances between residues with long sequence separation length still remains challenging.</p>
<p>In this paper, we propose DuetDis, which uses duet deep learning models for distance prediction. DuetDis adopts two complementary feature sets, one set is mainly composed of 2D coevolutionary couplings, and another set contains mainly 1D sequence-based features. We trained 10 sub-models using two different networks (Net1 and Net2), two different sets of features (FeatSet1 and FeatSet2), and four different MSAs (MSA_All, MSA_Top, MSA_1, MSA_2). By evaluating 10 sub-models based on the large-scale test set, we found that: 1) prediction results from different feature sets show obvious differences; 2) ensembling different feature sets can improve the prediction performance; and 3) high-quality MSA used for both training and testing can greatly improve the prediction performance. DuetDis is also compared with 11 widely used contact/distance predictors. The experimental results show that DuetDis outperforms the peer methods in terms of overall prediction precisions, model reliability, and robustness against shallow MSA.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>HZ, YH, and ZB conducted the experiments; all authors analyzed the data; HZ and WX wrote the manuscript.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This work was partly supported by the National Key Research and Development Program of China under Grant No. 2018YFB0204403, Strategic Priority CAS Project XDB38050100, the Key Research and Development Project of Guangdong Province under Grant No. 2021B0101310002, National Science Foundation of China under Grant No. U1813203, the Shenzhen Basic Research Fund under Grant Nos. RCYX2020071411473419, JCYJ20200109114818703, and JSGG20201102163800001, and CAS Key Lab under Grant No. 2011DP173015.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adhikari</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A Fully Open-Source Framework for Deep Learning Protein Real-Valued Distances</article-title>. <source>Sci. Rep.</source> <volume>10</volume> (<issue>1</issue>), <fpage>13374</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-70181-0</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adhikari</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bhattacharya</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>CONFOLD: Residue-Residue Contact-Guidedab Initioprotein Folding</article-title>. <source>Proteins</source> <volume>83</volume> (<issue>8</issue>), <fpage>1436</fpage>&#x2013;<lpage>1449</lpage>. <pub-id pub-id-type="doi">10.1002/prot.24829</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adhikari</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>DNCON2: Improved Protein Contact Prediction Using Two-Level Deep Convolutional Neural Networks</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>9</issue>), <fpage>1466</fpage>&#x2013;<lpage>1472</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx781</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altschul</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Madden</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Sch&#xe4;ffer</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>1997</year>). <article-title>Gapped BLAST and PSI-BLAST: a New Generation of Protein Database Search Programs</article-title>. <source>Nucleic Acids Res.</source> <volume>25</volume> (<issue>17</issue>), <fpage>3389</fpage>&#x2013;<lpage>3402</lpage>. <pub-id pub-id-type="doi">10.1093/nar/25.17.3389</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anfinsen</surname>
<given-names>C. B.</given-names>
</name>
</person-group> (<year>1973</year>). <article-title>Principles that Govern the Folding of Protein Chains</article-title>. <source>Science</source> <volume>181</volume> (<issue>4096</issue>), <fpage>223</fpage>&#x2013;<lpage>230</lpage>. <pub-id pub-id-type="doi">10.1126/science.181.4096.223</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anishchenko</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Pellock</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Chidyausiku</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Ramelot</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Ovchinnikov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>De Novo protein Design by Deep Network Hallucination</article-title>. <source>Nature</source> <volume>600</volume> (<issue>7889</issue>), <fpage>547</fpage>&#x2013;<lpage>552</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-04184-w</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Asz&#xf3;di</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>W. R.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Homology Modelling by Distance Geometry</article-title>. <source>Folding Des.</source> <volume>1</volume> (<issue>5</issue>), <fpage>325</fpage>&#x2013;<lpage>334</lpage>. </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baldassi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zamparo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Feinauer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Procaccini</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zecchina</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Weigt</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Fast and Accurate Multivariate Gaussian Modeling of Protein Families: Predicting Residue Contacts and Protein-Interaction Partners</article-title>. <source>PloS one</source> <volume>9</volume> (<issue>3</issue>), <fpage>e92721</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0092721</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Betancourt</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Thirumalai</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Pair Potentials for Protein Folding: Choice of Reference States and Sensitivity of Predicted Native States to Variations in the Interaction Schemes</article-title>. <source>Protein Sci.</source> <volume>8</volume> (<issue>2</issue>), <fpage>361</fpage>&#x2013;<lpage>369</lpage>. <pub-id pub-id-type="doi">10.1110/ps.8.2.361</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Baldi</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Improved Residue Contact Prediction Using Support Vector Machines and a Large Feature Set</article-title>. <source>Bmc Bioinformatics</source> <volume>8</volume> (<issue>1</issue>), <fpage>113</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-8-113</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cong</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Anishchenko</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Ovchinnikov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Protein Interaction Networks Revealed by Proteome Coevolution</article-title>. <source>Science</source> <volume>365</volume> (<issue>6449</issue>), <fpage>185</fpage>&#x2013;<lpage>189</lpage>. <pub-id pub-id-type="doi">10.1126/science.aaw6718</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Predicting the Real&#x2010;Valued Inter&#x2010;Residue Distances for Proteins</article-title>. <source>Adv. Sci.</source> <volume>7</volume> (<issue>19</issue>), <fpage>2001314</fpage>. <pub-id pub-id-type="doi">10.1002/advs.202001314</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>DeepConPred2: An Improved Method for the Prediction of Protein Residue Contacts</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>16</volume>, <fpage>503</fpage>&#x2013;<lpage>510</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2018.10.009</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Du</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Prediction of Residue-Residue Contact Matrix for Protein-Protein Interaction with Fisher Score Features and Deep Learning</article-title>. <source>Methods</source> <volume>110</volume>, <fpage>97</fpage>&#x2013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymeth.2016.06.001</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dunn</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Wahl</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Gloor</surname>
<given-names>G. B.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Mutual Information without the Influence of Phylogeny or Entropy Dramatically Improves Residue Contact Prediction</article-title>. <source>Bioinformatics</source> <volume>24</volume> (<issue>3</issue>), <fpage>333</fpage>&#x2013;<lpage>340</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btm604</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ekeberg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>L&#xf6;vkvist</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Weigt</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Aurell</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Improved Contact Prediction in Proteins: Using Pseudolikelihoods to Infer Potts Models</article-title>. <source>Phys. Rev. E Stat. Nonlin Soft Matter Phys.</source> <volume>87</volume> (<issue>1</issue>), <fpage>012707</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevE.87.012707</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Skolnick</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>DESTINI: A Deep-Learning Approach to Contact-Driven Protein Structure Prediction</article-title>. <source>Sci. Rep.</source> <volume>9</volume> (<issue>1</issue>), <fpage>3514</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-019-40314-1</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Greener</surname>
<given-names>J. G.</given-names>
</name>
<name>
<surname>Kandathil</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>D. T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep Learning Extends De Novo Protein Modelling Coverage of Genomes Using Iteratively Predicted Structural Constraints</article-title>. <source>Nat. Commun.</source> <volume>10</volume> (<issue>1</issue>), <fpage>3977</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-11994-0</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hanson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Paliwal</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Litfin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Accurate Prediction of Protein Contact Maps by Coupling Residual Two-Dimensional Bidirectional Long Short-Term Memory with Convolutional Neural Networks</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>23</issue>), <fpage>4039</fpage>&#x2013;<lpage>4045</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty481</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hanson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Paliwal</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Litfin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Improving Prediction of Protein Secondary Structure, Backbone Angles, Solvent Accessibility and Contact Numbers by Using Predicted Contact Maps and an Ensemble of Recurrent and Residual Convolutional Neural Networks</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>14</issue>), <fpage>2403</fpage>&#x2013;<lpage>2410</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty1006</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mortuza</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>H.-B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>NeBcon: Protein Contact Map Prediction Using Neural Network Training Coupled with Na&#xef;ve Bayes Classifiers</article-title>. <source>Bioinformatics</source> <volume>33</volume> (<issue>15</issue>), <fpage>2296</fpage>&#x2013;<lpage>2306</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx164</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Deep Residual Learning for Image Recognition</article-title>,&#x201d;in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <conf-date>17-19 June 1997</conf-date>. <conf-loc>Juan, PR, USA</conf-loc>. (<publisher-name>IEEE</publisher-name>). <pub-id pub-id-type="doi">10.1109/cvpr.2016.90</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jain</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Terashi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kagaya</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Venkata Subramaniya</surname>
<given-names>S. R. M.</given-names>
</name>
<name>
<surname>Christoffer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kihara</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Analyzing Effect of Quadruple Multiple Sequence Alignments on Deep Learning Based Protein Inter-residue Distance Prediction</article-title>. <source>Scientific Rep.</source> <volume>11</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-87204-z</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johnson</surname>
<given-names>L. S.</given-names>
</name>
<name>
<surname>Eddy</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Portugaly</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Hidden Markov Model Speed Heuristic and Iterative HMM Search Procedure</article-title>. <source>BMC bioinformatics</source> <volume>11</volume> (<issue>1</issue>), <fpage>431</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-11-431</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jones</surname>
<given-names>D. T.</given-names>
</name>
<name>
<surname>Buchan</surname>
<given-names>D. W. A.</given-names>
</name>
<name>
<surname>Cozzetto</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Pontil</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>PSICOV: Precise Structural Contact Prediction Using Sparse Inverse Covariance Estimation on Large Multiple Sequence Alignments</article-title>. <source>Bioinformatics</source> <volume>28</volume> (<issue>2</issue>), <fpage>184</fpage>&#x2013;<lpage>190</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr638</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jones</surname>
<given-names>D. T.</given-names>
</name>
<name>
<surname>Kandathil</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>High Precision in Protein Contact Prediction Using Fully Convolutional Neural Networks and Minimal Sequence Features</article-title>. <source>Bioinformatics</source> <volume>34</volume> (<issue>19</issue>), <fpage>3308</fpage>&#x2013;<lpage>3315</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty341</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jones</surname>
<given-names>D. T.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kosciolek</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tetchner</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>MetaPSICOV: Combining Coevolution Methods for Accurate Prediction of Contacts and Long Range Hydrogen Bonding in Proteins</article-title>. <source>Bioinformatics</source> <volume>31</volume> (<issue>7</issue>), <fpage>999</fpage>&#x2013;<lpage>1006</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btu791</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ju</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>An Efficient Greedy Incremental Sequence Clustering Algorithm</article-title>,&#x201d; in <source>International Symposium on Bioinformatics Research and Applications</source> (<publisher-name>Springer</publisher-name>, <publisher-loc>Cham</publisher-loc>). <pub-id pub-id-type="doi">10.1007/978-3-030-91415-8_50</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaj&#xe1;n</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hopf</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Kala&#x161;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Marks</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Rost</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>FreeContact: Fast and Free Software for Protein Contact Prediction from Residue Co-evolution</article-title>. <source>BMC bioinformatics</source> <volume>15</volume> (<issue>1</issue>), <fpage>85</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-15-85</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kamisetty</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ovchinnikov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Assessing the Utility of Coevolution-Based Residue-Residue Contact Predictions in a Sequence- and Structure-Rich Era</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>110</volume> (<issue>39</issue>), <fpage>15674</fpage>&#x2013;<lpage>15679</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1314045110</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kukic</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mirabello</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tradigo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Walsh</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Veltri</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Pollastri</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Toward an Accurate Prediction of Inter-residue Distances in Proteins Using 2D Recursive Neural Networks</article-title>. <source>BMC bioinformatics</source> <volume>15</volume> (<issue>1</issue>), <fpage>6</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-15-6</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>B.-C.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>A New Method for Revealing Correlated Mutations under the Structural and Functional Constraints in Proteins</article-title>. <source>Bioinformatics</source> <volume>25</volume> (<issue>19</issue>), <fpage>2506</fpage>&#x2013;<lpage>2513</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp455</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Study of Real-Valued Distance Prediction for Protein Structure Prediction with Deep Learning</article-title>. <source>Bioinformatics</source> <volume>37</volume> (<issue>19</issue>), <fpage>3197</fpage>&#x2013;<lpage>3203</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab333</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>D.-J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>ResPRE: High-Accuracy Protein Contact Prediction by Coupling Precision Matrix with Deep Residual Neural Networks</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>22</issue>), <fpage>4647</fpage>&#x2013;<lpage>4655</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz291</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bell</surname>
<given-names>E. W.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>D.-J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Deducing High-Accuracy Protein Contact-Maps from a Triplet of Coevolutionary Matrices through Deep Residual Convolutional Networks</article-title>. <source>Plos Comput. Biol.</source> <volume>17</volume> (<issue>3</issue>), <fpage>e1008865</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1008865</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Palmedo</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Berger</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Enhancing Evolutionary Couplings with Deep Convolutional Neural Networks</article-title>. <source>Cell Syst.</source> <volume>6</volume> (<issue>1</issue>), <fpage>65</fpage>&#x2013;<lpage>74</lpage>. <comment>e3</comment>. <pub-id pub-id-type="doi">10.1016/j.cels.2017.11.014</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Malinin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gales</surname>
<given-names>M. J. F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Uncertainty Estimation in Autoregressive Structured Prediction</article-title>. <conf-name>9th International Conference on Learning Representations, {ICLR} 2021, Virtual Event</conf-name>, <conf-loc>Austria</conf-loc>, <conf-date>May 3-7, 2021</conf-date>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=jN5y-zb5Q7m">https://openreview.net/forum?id&#x003D;jN5y-zb5Q7m</ext-link>
</comment>. </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marks</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Hopf</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Protein Structure Prediction from Sequence Variation</article-title>. <source>Nat. Biotechnol.</source> <volume>30</volume> (<issue>11</issue>), <fpage>1072</fpage>&#x2013;<lpage>1080</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.2419</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McAllister</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Floudas</surname>
<given-names>C. A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>&#x3b1;-Helical Topology Prediction and Generation of Distance Restraints in Membrane Proteins</article-title>. <source>Biophysical J.</source> <volume>95</volume> (<issue>11</issue>), <fpage>5281</fpage>&#x2013;<lpage>5295</lpage>. <pub-id pub-id-type="doi">10.1529/biophysj.108.132241</pub-id> </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Michel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hayat</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Skwark</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Marks</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Elofsson</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>PconsFold: Improved Contact Predictions Improve Protein Models</article-title>. <source>Bioinformatics</source> <volume>30</volume> (<issue>17</issue>), <fpage>i482</fpage>&#x2013;<lpage>i488</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btu458</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Michel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Men&#xe9;ndez Hurtado</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Elofsson</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>PconsC4: Fast, Accurate and Hassle-free Contact Predictions</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>15</issue>), <fpage>2677</fpage>&#x2013;<lpage>2679</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty1036</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morcos</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Pagnani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lunt</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bertolino</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Marks</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Sander</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Direct-coupling Analysis of Residue Coevolution Captures Native Contacts across many Protein Families</article-title>. <source>Proc. Natl. Acad. Sci. U S A.</source> <volume>108</volume> (<issue>49</issue>), <fpage>E1293</fpage>&#x2013;<lpage>E1301</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1111471108</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pollock</surname>
<given-names>D. D.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>W. R.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Effectiveness of Correlation Analysis in Identifying Protein Residues Undergoing Correlated Evolution</article-title>. <source>Protein Eng. Des. Selection</source> <volume>10</volume> (<issue>6</issue>), <fpage>647</fpage>&#x2013;<lpage>657</lpage>. <pub-id pub-id-type="doi">10.1093/protein/10.6.647</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rahman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Newton</surname>
<given-names>M. A. H.</given-names>
</name>
<name>
<surname>Islam</surname>
<given-names>M. K. B.</given-names>
</name>
<name>
<surname>Sattar</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Enhancing Protein Inter-residue Real Distance Prediction by Scrutinising Deep Learning Models</article-title>. <source>Sci. Rep.</source> <volume>12</volume> (<issue>1</issue>), <fpage>787</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-04441-y</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rajgaria</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>McAllister</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Floudas</surname>
<given-names>C. A.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Towards Accurate Residue-Residue Hydrophobic Contact Prediction for &#x3b1; Helical Proteins via Integer Linear Optimization</article-title>. <source>Proteins</source> <volume>74</volume> (<issue>4</issue>), <fpage>929</fpage>&#x2013;<lpage>947</lpage>. <pub-id pub-id-type="doi">10.1002/prot.22202</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rajgaria</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Floudas</surname>
<given-names>C. A.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Contact Prediction for Beta and Alpha-Beta Proteins Using Integer Linear Optimization and its Impact on the First Principles 3D Structure Prediction Method ASTRO-FOLD</article-title>. <source>Proteins</source> <volume>78</volume> (<issue>8</issue>), <fpage>1825</fpage>&#x2013;<lpage>1846</lpage>. <pub-id pub-id-type="doi">10.1002/prot.22696</pub-id> </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Remmert</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Biegert</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hauser</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>S&#xf6;ding</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>HHblits: Lightning-Fast Iterative Protein Sequence Searching by HMM-HMM Alignment</article-title>. <source>Nat. Methods</source> <volume>9</volume> (<issue>2</issue>), <fpage>173</fpage>&#x2013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.1818</pub-id> </citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reza</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hossain</surname>
<given-names>M. T.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>COMTOP: Protein Residue-Residue Contact Prediction through Mixed Integer Linear Optimization</article-title>. <source>Membranes</source> <volume>11</volume> (<issue>7</issue>), <fpage>503</fpage>. <pub-id pub-id-type="doi">10.3390/membranes11070503</pub-id> </citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schlessinger</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Punta</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rost</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Natively Unstructured Regions in Proteins Identified from Contact Predictions</article-title>. <source>Bioinformatics</source> <volume>23</volume> (<issue>18</issue>), <fpage>2376</fpage>&#x2013;<lpage>2384</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btm349</pub-id> </citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Seemayer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gruber</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>S&#xf6;ding</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>CCMpred-fast and Precise Prediction of Protein Residue-Residue Contacts from Correlated Mutations</article-title>. <source>Bioinformatics</source> <volume>30</volume> (<issue>21</issue>), <fpage>3128</fpage>&#x2013;<lpage>3130</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btu500</pub-id> </citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Senior</surname>
<given-names>A. W.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Jumper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kirkpatrick</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sifre</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Improved Protein Structure Prediction Using Potentials from Deep Learning</article-title>. <source>Nature</source> <volume>577</volume>, <fpage>706</fpage>&#x2013;<lpage>710</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-019-1923-7</pub-id> </citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shimomura</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Nishijima</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kikuchi</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A New Technique for Predicting Intrinsically Disordered Regions Based on Average Distance Map Constructed with Inter-residue Average Distance Statistics</article-title>. <source>BMC Struct. Biol.</source> <volume>19</volume> (<issue>1</issue>), <fpage>3</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1186/s12900-019-0101-3</pub-id> </citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Litfin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Paliwal</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>SPOT-Contact-LM: Improving Single-Sequence-Based Prediction of Protein Contact Map Using a Transformer Language Model</article-title>. <source>Bioinformatics</source>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btac053</pub-id> </citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Skwark</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Abdel-Rehim</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Elofsson</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>PconsC: Combination of Direct Information Methods and Alignments Improves Contact Prediction</article-title>. <source>Bioinformatics</source> <volume>29</volume> (<issue>14</issue>), <fpage>1815</fpage>&#x2013;<lpage>1816</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt259</pub-id> </citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>M. M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Improved Protein Structure Prediction Using a New Multi&#x2010;Scale Network and Homologous Templates</article-title>. <source>Adv. Sci.</source> <volume>8</volume>, <fpage>2102592</fpage>. <pub-id pub-id-type="doi">10.1002/advs.202102592</pub-id> </citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tegge</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Eickholt</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>NNcon: Improved Protein Contact Map Prediction Using 2D-Recursive Neural Networks</article-title>. <source>Nucleic Acids Res.</source> <volume>37</volume>, <fpage>W515</fpage>&#x2013;<lpage>W518</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkp305</pub-id> </citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vangone</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bonvin</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Contacts-based Prediction of Binding Affinity in Protein-Protein Complexes</article-title>. <source>elife</source> <volume>4</volume>, <fpage>e07454</fpage>. <pub-id pub-id-type="doi">10.7554/eLife.07454</pub-id> </citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Walsh</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Ba&#xf9;</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Martin</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Mooney</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Vullo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pollastri</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Ab Initio and Template-Based Prediction of Multi-Class Distance Maps by Two-Dimensional Recursive Neural Networks</article-title>. <source>BMC Struct. Biol.</source> <volume>9</volume> (<issue>1</issue>), <fpage>5</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1186/1472-6807-9-5</pub-id> </citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Accurate De Novo Prediction of Protein Contact Map by Ultra-deep Learning Model</article-title>. <source>Plos Comput. Biol.</source> <volume>13</volume> (<issue>1</issue>), <fpage>e1005324</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005324</pub-id> </citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Predicting Protein Contact Map Using Evolutionary and Physical Constraints by Integer Programming</article-title>. <source>Bioinformatics</source> <volume>29</volume> (<issue>13</issue>), <fpage>i266</fpage>&#x2013;<lpage>i273</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt211</pub-id> </citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Floudas</surname>
<given-names>C. A.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Enhanced Inter-helical Residue Contact Prediction in Transmembrane Proteins</article-title>. <source>Chem. Eng. Sci.</source> <volume>66</volume> (<issue>19</issue>), <fpage>4356</fpage>&#x2013;<lpage>4369</lpage>. <pub-id pub-id-type="doi">10.1016/j.ces.2011.04.033</pub-id> </citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weigt</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Szurmant</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hoch</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Hwa</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Identification of Direct Residue Contacts in Protein-Protein Interaction by Message Passing</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>106</volume> (<issue>1</issue>), <fpage>67</fpage>&#x2013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0805923106</pub-id> </citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Anishchenko</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Cong</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Protein Contact Prediction Using Metagenome Sequence Data and Residual Neural Networks</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>1</issue>), <fpage>41</fpage>&#x2013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz477</pub-id> </citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>A Comprehensive Assessment of Sequence-Based and Template-Based Methods for Protein Contact Prediction</article-title>. <source>Bioinformatics</source> <volume>24</volume> (<issue>7</issue>), <fpage>924</fpage>&#x2013;<lpage>931</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btn069</pub-id> </citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>DeepDist: Real-Value Inter-Residue Distance Prediction with Deep Residual Convolutional Network</article-title>. <source>BMC Bioinform.</source> <volume>22</volume>, <fpage>30</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-021-04269-3</pub-id> </citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Distance-based Protein Folding Powered by Deep Learning</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>116</volume> (<issue>34</issue>), <fpage>16856</fpage>&#x2013;<lpage>16865</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1821309116</pub-id> </citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Anishchenko</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ovchinnikov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Improved Protein Structure Prediction Using Predicted Interresidue Orientations</article-title>. <source>Proc Natl Acad Sci U S A.</source> <volume>117</volume>(<issue>3</issue>), <fpage>1496</fpage>-<lpage>1503</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1914677117</pub-id> </citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bei</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ju</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Saravanan</surname>
<given-names>K. M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Evaluation of Residue-Residue Contact Prediction Methods: From Retrospective to Prospective</article-title>. <source>Plos Comput. Biol.</source> <volume>17</volume> (<issue>5</issue>), <fpage>e1009027</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1009027</pub-id> </citation>
</ref>
<ref id="B69">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ting</surname>
<given-names>H. F.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Protein Interresidue Contact Prediction Based on Deep Learning and Massive Features from Multi-Sequence Alignment</article-title>,&#x201d; in <conf-name>International Conference on Parallel and Distributed Computing: Applications and Technologies</conf-name>, <conf-loc>Shenzhen, China</conf-loc>, <conf-date>December 28&#x2010;30</conf-date> (<publisher-loc>Shenzhen</publisher-loc>: <publisher-name>Springer</publisher-name>). </citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ting</surname>
<given-names>H.-F.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xi</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Protein Residue Contact Prediction Based on Deep Learning and Massive Statistical Features from Multi-Sequence Alignment</article-title>. <source>Tsinghua Sci. Technol.</source> <volume>27</volume> (<issue>5</issue>), <fpage>843</fpage>&#x2013;<lpage>854</lpage>. <pub-id pub-id-type="doi">10.26599/tst.2021.9010064</pub-id> </citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Bei</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Floudas</surname>
<given-names>C. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>COMSAT: Residue Contact Prediction of Transmembrane Proteins Based on Support Vector Machines and Mixed Integer Linear Programming</article-title>. <source>Proteins</source> <volume>84</volume> (<issue>3</issue>), <fpage>332</fpage>&#x2013;<lpage>348</lpage>. <pub-id pub-id-type="doi">10.1002/prot.24979</pub-id> </citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Hansmann</surname>
<given-names>U. H. E.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Fibril-Barrel Transitions in Cylindrin Amyloids</article-title>. <source>J. Chem. Theor. Comput.</source> <volume>13</volume> (<issue>8</issue>), <fpage>3936</fpage>&#x2013;<lpage>3944</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jctc.7b00383</pub-id> </citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>A Position-specific Distance-dependent Statistical Potential for Protein Structure and Functional Study</article-title>. <source>Structure</source> <volume>20</volume> (<issue>6</issue>), <fpage>1118</fpage>&#x2013;<lpage>1126</lpage>. <pub-id pub-id-type="doi">10.1016/j.str.2012.04.003</pub-id> </citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wuyun</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Pearce</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>FUpred: Detecting Protein Domains through Deep-Learning-Based Contact Map Prediction</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>12</issue>), <fpage>3749</fpage>&#x2013;<lpage>3757</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa217</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>