<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioinform.</journal-id>
<journal-title>Frontiers in Bioinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioinform.</abbrev-journal-title>
<issn pub-type="epub">2673-7647</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1477909</article-id>
<article-id pub-id-type="doi">10.3389/fbinf.2024.1477909</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioinformatics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>DCMA: faster protein backbone dihedral angle prediction using a dilated convolutional attention-based neural network</article-title>
<alt-title alt-title-type="left-running-head">Zhang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbinf.2024.1477909">10.3389/fbinf.2024.1477909</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhang</surname>
<given-names>Buzhong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2808292/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zheng</surname>
<given-names>Meili</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Yuzhou</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Quan</surname>
<given-names>Lijun</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1516807/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Computer and Information</institution>, <institution>Anqing Normal University</institution>, <addr-line>Anqing</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Jiangsu Provincial Key Laboratory for Computer Information Processing Technology</institution>, <institution>Soochow University</institution>, <addr-line>Suzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Information Engineering</institution>, <institution>Nanjing Xiaozhuang University</institution>, <addr-line>Nanjing</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>School of Computer Science and Technology</institution>, <institution>Soochow University</institution>, <addr-line>Suzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1710191/overview">Yaan J. Jang</ext-link>, University of Oxford, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1310188/overview">Fabien Plisson</ext-link>, Center for Research and Advanced Studies (CINVESTAV), Mexico</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2815822/overview">Guijun Zhang</ext-link>, Zhejiang University of Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Buzhong Zhang, <email>zhbzhong@aqnu.edu.cn</email>; Lijun Quan, <email>ljquan@suda.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>18</day>
<month>10</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>4</volume>
<elocation-id>1477909</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>08</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>09</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Zhang, Zheng, Zhang and Quan.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Zhang, Zheng, Zhang and Quan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The dihedral angle of the protein backbone can describe the main structure of the protein, which is of great significance for determining the protein structure. Many computational methods have been proposed to predict this critically important protein structure, including deep learning. However, these heavyweight methods require more computational resources, and the training time becomes intolerable. In this article, we introduce a novel lightweight method, named dilated convolution and multi-head attention (DCMA), that predicts protein backbone torsion dihedral angles <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. DCMA is stacked by five layers of two hybrid inception blocks and one multi-head attention block (I2A1) module. The hybrid inception blocks consisting of multi-scale convolutional neural networks and dilated convolutional neural networks are designed for capturing local and long-range sequence-based features. The multi-head attention block supplementally strengthens this operation. The proposed DCMA is validated on public critical assessment of protein structure prediction (CASP) benchmark datasets. Experimental results show that DCMA obtains better or comparable generalization performance. Compared to best-so-far methods, which are mostly ensemble models and constructed of recurrent neural networks, DCMA is an individual model that is more lightweight and has a shorter training time. The proposed model could be applied as an alternative method for predicting other protein structural features.</p>
</abstract>
<kwd-group>
<kwd>protein dihedral angles</kwd>
<kwd>lightweight model</kwd>
<kwd>dilated convolution</kwd>
<kwd>multi-head attention</kwd>
<kwd>hybrid inception blocks</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Protein Bioinformatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Proteins play important roles in biological activities and often fold into unique three-dimensional structures to perform their biological functions. However, experimentally determining protein tertiary structures is costly and time consuming. Predicting protein tertiary structures from their corresponding sequences is still a challenging problem in computational biology. An integral part of predicting tertiary structures is to predict interval structural properties, such as secondary structures, solvent-accessible surface area, backbone dihedral angles, and contact maps. The backbone structure of a protein can be described continuously by backbone dihedral angles <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The backbone torsion angle prediction is beneficial for protein structure prediction. The dihedral angle prediction has many applications in protein structure prediction, including (i) better secondary structure prediction, (ii) generation of multiple sequence alignments, (iii) identification of protein folds, and (iv) fragment-free tertiary structure prediction (<xref ref-type="bibr" rid="B22">Singh et al., 2014</xref>). Many computational methods, especially deep learning-based models, have been applied in this field.</p>
<p>In 1993, a discrete approach was used to predict backbone dihedral angles that removed any approximations, including the assumption that the effects of adjacent residues were uncorrelated (<xref ref-type="bibr" rid="B17">Kang et al., 1993</xref>). In 2005, a continuous neural network-based method was proposed to predict protein secondary structure and backbone dihedral angles (<xref ref-type="bibr" rid="B36">Wood and Hirst, 2005</xref>). Other machine learning methods have also been applied to the prediction of protein dihedral angles, such as ANGLOR (<xref ref-type="bibr" rid="B37">Wu and Zhang, 2008</xref>) and TANGLE (<xref ref-type="bibr" rid="B25">Song et al., 2012</xref>) using support vector machines and neural networks, TALOS&#x2b; (<xref ref-type="bibr" rid="B21">Shen et al., 2009</xref>), SPINE X, and Real-SPINE3.0 using neural networks, DANGLE (<xref ref-type="bibr" rid="B3">Cheung et al., 2010</xref>) using Bayesian, conditional random field (<xref ref-type="bibr" rid="B45">Zhang et al., 2013</xref>), and so on.</p>
<p>In recent years, deep learning methods have been successfully applied to the prediction of protein structural properties, including protein backbone dihedral angles. A deep recurrent restricted Boltzmann machine (DReRBM) was developed to research protein dihedral angles (<xref ref-type="bibr" rid="B19">Li et al., 2017</xref>). RaptorX-angle combines K-means clustering and deep learning techniques to predict dihedral angles (<xref ref-type="bibr" rid="B7">Gao et al., 2018</xref>). Spider 3 (<xref ref-type="bibr" rid="B12">Heffernan et al., 2017</xref>), which eliminates the effect of the sliding window, used the machine learning model of the bidirectional long short-term memory (BLSTM) (<xref ref-type="bibr" rid="B13">Hochreiter and Schmidhuber, 1997</xref>) recurrent neural network (<xref ref-type="bibr" rid="B20">Schuster and Paliwal, 1997</xref>; <xref ref-type="bibr" rid="B8">Graves et al., 2014</xref>). The DeepRIN (<xref ref-type="bibr" rid="B6">Fang et al., 2018b</xref>) was designed based on the combination of the inception (<xref ref-type="bibr" rid="B30">Szegedy et al., 2016</xref>) and the ResNet (<xref ref-type="bibr" rid="B11">He et al., 2016</xref>) networks. SPOT-1D used an ensemble of BLSTM and ResNet to improve the prediction of protein secondary structure, backbone dihedral angles, solvent accessibility, etc. (<xref ref-type="bibr" rid="B10">Hanson et al., 2019</xref>). SPOT-1D integrated three LSTM models, three LSTMResNet models, and three ResNet-LSTM models and integrated contact maps as model input and to boost its performance. Klausen and colleagues proposed the NetSurfP-2.0 model, which used an architecture consisting of a convolutional neural network (CNN) and LSTM Networks (<xref ref-type="bibr" rid="B18">Klausen et al., 2019</xref>). Xu and colleagues proposed OPUS-TASS, a protein backbone dihedral angles and secondary structure predictor (<xref ref-type="bibr" rid="B39">Xu et al., 2020</xref>). It is an ensemble model; its individual model parts consist of CNN, LSTM, and modified transformer networks. Zhang and colleagues proposed CRRNN2, which introduced a multi-task deep learning method based on BRNNs, one-dimensional (1D) CNN, and an inception network, which can concurrently predict protein secondary structure, solvent accessibility, and backbone dihedral angles (<xref ref-type="bibr" rid="B44">Zhang et al., 2021</xref>). As an upgraded version of OPUS-TASS, OPUS-TASS2 integrated global structure information generated by trRosetta (<xref ref-type="bibr" rid="B4">Du et al., 2021</xref>) and achieves SOTA performance. OPUS-TASS2 adopts an ensemble strategy as OPUS-TASS and SPOT-1D, and it consists of nine models (<xref ref-type="bibr" rid="B40">Xu et al., 2022</xref>).</p>
<p>Recently, AlphaFold2 has achieved great success in predicting protein monomer structures (<xref ref-type="bibr" rid="B16">Jumper et al., 2021</xref>; <xref ref-type="bibr" rid="B14">Ismi et al., 2022</xref>). However, accuracy for single-sequence-based prediction of secondary structures is far from the theoretical limit of 86%&#x2013;90% (<xref ref-type="bibr" rid="B46">Zhou et al., 2023</xref>). The bottleneck resides in the immense computational demands of running the AlphaFold2 model, both in terms of computing power and runtime. Therefore, there is still a need for prediction tools that can predict protein backbone angles in a faster and more accurate manner. A recurrent neural network (RNN) maintains a vector of activations for each timestep that can remember prior input to influence the current input and output. RNNs can be easily used for sequential or time series data (<xref ref-type="bibr" rid="B15">Jozefowicz et al., 2015</xref>). The best-so-far deep learning methods are essentially constructed by RNNs like SPOT-1D (<xref ref-type="bibr" rid="B10">Hanson et al., 2019</xref>), OPUS-TASS (<xref ref-type="bibr" rid="B39">Xu et al., 2020</xref>), OPUS-TASS2 (<xref ref-type="bibr" rid="B40">Xu et al., 2022</xref>), NetSurfP-2.0 (<xref ref-type="bibr" rid="B18">Klausen et al., 2019</xref>), and CRRNN2 (<xref ref-type="bibr" rid="B44">Zhang et al., 2021</xref>). Because the computation of each step in an RNN depends on the previous step, the recurrent computations are less amenable to parallelization. In contrast to the models built by convolution or attention networks, RNN-based models need more training and running times. Moreover, these heavyweight models consume more computing resources, which is not conducive to training or inference.</p>
<p>In this article, we designed a new hybrid inception block consisting of 1D CNNs and dilated CNNs (<xref ref-type="bibr" rid="B41">Yu and Koltun, 2016</xref>). As an alternative to an RNN, a novel architecture with two hybrid inception blocks and one multi-head attention (<xref ref-type="bibr" rid="B33">Vaswani et al., 2017</xref>) block called an I2A1 module is intended to capture local and long-range features, which are comprised of two hybrid inception blocks and augmented by one multi-head attention network. The dilated convolution and multi-head attention (DCMA) novel protein backbone dihedral angle predictor is mainly constructed from I2A1 modules. Hence, we have made the following outstanding contributions: 1) proposed a faster method that can substitute for RNNs and offers comparable performance, and 2) provided a more lightweight tool for predicting dihedral angles that is more friendly to biological or medical researchers.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Datasets</title>
<p>We used the same training (<xref ref-type="bibr" rid="B10">Hanson et al., 2019</xref>) and validation sets as SPOT-1D and OPUS-TASS 1/2 for a fair comparison with most state-of-the-art methods. The sequences were culled from the PISCES server (<xref ref-type="bibr" rid="B34">Wang et al., 2003</xref>) by SPOT-1D in February 2017, with the following constraints: resolution <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mo>&#x3e;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>2.5 &#xc5;, R-free <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mo>&#x3c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 1, a sequence identity truncation rate of 25%, and the sequence length <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mo>&#x2264;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 700. Finally, the training set and validation set contain 10,029 proteins and 983 proteins, respectively.</p>
<p>To evaluate the performance of different methods, we performed the method on six public independent test sets: (1) The CASP12 dataset contains 40 proteins; (2) CASP13, which contains 32 proteins; (3) CASP-FM (56), collected by SAINT (<xref ref-type="bibr" rid="B32">Uddin et al., 2020</xref>), which contains 10 template free modeling (FM) targets from CASP13, 22 FM targets from CASP12, 16 FM targets from CASP11, and 8 FM targets from CASP10; (4) the CASP12-FM dataset, collected by <xref ref-type="bibr" rid="B23">Singh et al. (2021a</xref>), which contains 22 FM proteins from CASP12; (5) the CASP13-FM dataset, collected by <xref ref-type="bibr" rid="B23">Singh et al. (2021a</xref>), which contains 17 FM proteins from CASP13; and the (6) CASP14-FM dataset, collected by <xref ref-type="bibr" rid="B40">Xu et al. (2022)</xref>, which contains 15 FM proteins from CASP14.</p>
</sec>
<sec id="s2-2">
<title>2.2 Input features</title>
<p>DCMA takes three groups of sequence-based features as input: a position-specific scoring matrix (PSSM) profile, a hidden Markov model (HMM) profile, and residues coding. As SPOT-1D reported, each 20-dimensional protein PSSM was generated by three iterations of PSI-BLAST (<xref ref-type="bibr" rid="B1">Altschul et al., 1997</xref>) against the UniRef90 sequence database updated in April 2018. The 30-dimensional HMM sequence profiles are re-generated by HHBlits (v3.1.0) (<xref ref-type="bibr" rid="B26">Steinegger et al., 2019</xref>) with default parameters based on the UniRef30 database updated in June 2020. Similarly to CRRNN&#x2019;s schema (<xref ref-type="bibr" rid="B43">Zhang et al., 2018</xref>), a one-hot vector of residues coding is mapped to a 22-dimensional dense vector.</p>
</sec>
<sec id="s2-3">
<title>2.3 Outputs</title>
<p>To remove the effect of the angle&#x2019;s periodicity, we employ a pair of sine and cosine values for each torsion angle as the output instead of directly predicting <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> or <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. As a result, there are four outputs: <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mi>sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mi>cos</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:mi>sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mi>cos</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.The predicted angle <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is defined as</p>
<p>
<inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>cos</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<p>The multi-task learning strategy of predicting protein structural properties concurrently has been proven to be effective by CRRNN2 (<xref ref-type="bibr" rid="B44">Zhang et al., 2021</xref>). We also adopted the same multi-task learning schema as CRRNN2, in which the auxiliary output during the training period is protein secondary structure (Q3 and Q8) and solvent accessibility. The loss function and respective output ratio in DCMA are the same as in CRRNN2.</p>
</sec>
<sec id="s2-4">
<title>2.4 DCMA model</title>
<p>As illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>, our DCMA model consists of three parts: a pre-processing block, five stacked I2A1 modules, and two fully connected layers. The input features are transformed in the pre-processing block, and its structure is demonstrated in <xref ref-type="fig" rid="F2">Figure 2A</xref>. The I2A1 module is mainly constructed by two cascaded hybrid inception blocks and one multi-head attention block, as <xref ref-type="fig" rid="F2">Figure 2B</xref> shows. The details of these blocks will be introduced below.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Model architecture of the DCMA.</p>
</caption>
<graphic xlink:href="fbinf-04-1477909-g001.tif"/>
</fig>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Architecture of pre-processing block <bold>(A)</bold> and I2A1 module <bold>(B)</bold>.</p>
</caption>
<graphic xlink:href="fbinf-04-1477909-g002.tif"/>
</fig>
<sec id="s2-4-1">
<title>2.4.1 Pre-processing block</title>
<p>As <xref ref-type="fig" rid="F2">Figure 2A</xref> and <xref ref-type="disp-formula" rid="e1">Equation 1</xref> show, the representing residue features, including 20-dimension (D) PSSM, 22-D residue coding, and 30-D HMM, are aggregated and transformed into 256-D tensors by one dimension and one kernel (1D1) CNN (<xref ref-type="bibr" rid="B43">Zhang et al., 2018</xref>). The weight constraint of dropout (p &#x3d; 0.5) used to avoid overfitting was applied to the output of 1D1 CNN. Then, the tensors denoted as input are fed to each I2A1 module and the first dense layer.<disp-formula id="e1">
<mml:math id="m15">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>M</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>H</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>M</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mi>O</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn mathvariant="normal">1</mml:mn>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-4-2">
<title>2.4.2 Hybrid inception block</title>
<p>CNNs provide the property (<xref ref-type="bibr" rid="B27">Strubell et al., 2017</xref>) that parallelizes runtime independent of the sequence length maximizes GPU resource usage and minimizes the training and evaluating time. However, a CNN&#x2019;s perception is limited by the input size. As Strubell&#x2019;s description notes (<xref ref-type="bibr" rid="B27">Strubell et al., 2017</xref>), the maximum perception length r of CNN is expressed as <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the stacked layers number, and <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is filter size. And the receptive width is promoted to <inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, when <inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> dilated layers are stacked. Dilated convolution can cover a larger area of the input due to skipping some areas, and a large portion of the information is lost (<xref ref-type="bibr" rid="B35">Wang et al., 2018</xref>). A simple and useful solution is hybrid dilated convolutions, and this strategy has been proven practicable (<xref ref-type="bibr" rid="B35">Wang et al., 2018</xref>; <xref ref-type="bibr" rid="B38">Wu et al., 2016</xref>; <xref ref-type="bibr" rid="B2">Chen et al., 2017</xref>; <xref ref-type="bibr" rid="B24">Singh et al., 2021b</xref>).</p>
<p>Motivated by the effectiveness of the inception (<xref ref-type="bibr" rid="B29">Szegedy et al., 2015</xref>; <xref ref-type="bibr" rid="B30">Szegedy et al., 2016</xref>) network based on CNNs (<xref ref-type="bibr" rid="B6">Fang et al., 2018b</xref>; <xref ref-type="bibr" rid="B5">Fang et al., 2018a</xref>; <xref ref-type="bibr" rid="B32">Uddin et al., 2020</xref>), we proposed a newly hybrid inception block, as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. Four types of local features are aggregated from 1D CNNs with 64 filters and kernel size [1, 3, 5, 7] respectively, for the minimum length of protein secondary structures is three, and 1D CNN with kernel size one (1D1 CNN) is used to data dimension transformation. Another four groups of long-range features are perceived from hybrid dilation rates (d_rate) CNNs with 64 filters and kernel size 2. Similar to DeepLabv3&#x2019;s (<xref ref-type="bibr" rid="B2">Chen et al., 2017</xref>) configuration, dilation rates [2, 4, 8, 16] are applied. For better perceptive capability, four channels with stacked [1, 2, 3, 4] dilated CNNs, respectively, are combined. Multi-scale dilation rates { [2], [2, 4], [2, 4, 8], [2, 4, 8, 16]} are used in corresponding channels, respectively. When the eight parallel outputs are concatenated as a 512-dimensional tensor, the data dimension is transformed into 256 by a 1D1 CNN for model weight reduction.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Newly hybrid inception block. The 1D CNNs with 64 filters and kernel size [1, 3, 5, 7] are intended to capture sequence local futures. Dilated 1D CNNs with 64 filters, kernel size 2, and multi-scale dilation rates (d_rate) are used to capture sequence long-range dependencies.</p>
</caption>
<graphic xlink:href="fbinf-04-1477909-g003.tif"/>
</fig>
</sec>
</sec>
<sec id="s2-5">
<title>2.4.3 Multi-head attention</title>
<p>The attention mechanism (<xref ref-type="bibr" rid="B31">Tay et al., 2020</xref>) can be viewed as a graph-like inductive bias that connects all tokens in a sequence with a relevance-based pooling operation. Multi-head attention (<xref ref-type="bibr" rid="B33">Vaswani et al., 2017</xref>) allows the model to jointly attend to information from different representations and focus on different aspects of information. The advantage of attention is that it can capture long-term dependencies without being limited by sequence length. Because the result of each step does not depend on the previous step, steps can be done in parallel mode. We use a multi-head attention mechanism as a complement to the hybrid inception block. Eight heads are employed. In the DCMA model, we use multi-head attention as <xref ref-type="disp-formula" rid="e2">Equations 2</xref>&#x2013;<xref ref-type="disp-formula" rid="e4">4</xref>, the most popular attention in recent years, which combines multiple self-attention networks to divide the model into multiple heads to form multiple subspaces and can make models focus on different aspects of information. In our model, we employ eight heads, and tanh activated function is added to the output for smooth variation. In order to control model weight, the dimension of <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>head</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is reduced to 64, and the output dimension of the attention block is 256.<disp-formula id="e2">
<mml:math id="m22">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m23">
<mml:mrow>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2026;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>8</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math display="block" id="m24">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mfenced close=")" open="(" separators="none">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:msubsup>
<mml:mi>W</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
<mml:msubsup>
<mml:mi>W</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>K</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
<mml:msubsup>
<mml:mi>W</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>V</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">f</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mfenced close=")" open="(" separators="none">
<mml:mfrac>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:msubsup>
<mml:mi>W</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:msubsup>
<mml:mi>W</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>K</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
<mml:msqrt>
<mml:mi>d</mml:mi>
</mml:msqrt>
</mml:mfrac>
</mml:mfenced>
</mml:mrow>
<mml:mi>V</mml:mi>
<mml:msubsup>
<mml:mi>W</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>V</mml:mi>
</mml:msubsup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<sec id="s2-5-1">
<title>2.4.4 I2A1 module</title>
<p>The input data is fed parallel to one multi-headed attention block and two cascaded hybrid inception blocks, as <xref ref-type="fig" rid="F2">Figure 2B</xref> and <xref ref-type="disp-formula" rid="e5">Equations 5</xref>&#x2013;<xref ref-type="disp-formula" rid="e10">10</xref> show. The proposed module can effectively capture both the short-range and long-range dependencies. In the first I2A1 module (<italic>i</italic> &#x3d; 1), Input2 is the output of the pre_processing block, and Input1 is null. In other I2A1 modules, Input2 is the output of the previous I2A1 module, and Input1 is the output of the pre_processing block. For a better balance between the ability to model long-range dependencies and computational efficiency, only one attention block is joined with two cascaded hybrid inception blocks, and the dimension of concatenated data is also reduced from 1,024 (768, when in the first module) to 256 by 1D1 CNN.<disp-formula id="e5">
<mml:math id="m25">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>n</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mspace width="2.77695pt" class="tmspace"/>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mspace width="1em"/>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>b</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mspace width="2.77695pt" class="tmspace"/>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m26">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>g</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>b</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mspace width="2.77695pt" class="tmspace"/>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mspace width="1em"/>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>I</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>A</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>,</mml:mo>
<mml:mspace width="2.77695pt" class="tmspace"/>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m27">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m28">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="e9">
<mml:math id="m29">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>F</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m30">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn mathvariant="normal">1</mml:mn>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Experimental settings</title>
<p>The developed DCMA model was implemented in Keras, and the weights in DCMA were initialized using default values. The implementation was trained on a NVIDIA P6000 GPU. Adam optimization with an initial learning rate of 0.0004 was used to optimize the networks.</p>
<p>For training the model on GPU with batch input, proteins shorter than 700 AA are padded with all-zeros. Similar to the CRRNN2 experiment, the strategy of deep multi-task learning is also applied in the DCMA training period. In the inference period, only the backbone angle output is retained.</p>
</sec>
<sec id="s3-2">
<title>3.2 Evaluation metrics</title>
<p>To evaluate the predictive performance of protein backbone dihedral angles, the mean absolute error (MAE) (<xref ref-type="bibr" rid="B28">Sunghoon et al., 2011</xref>) and Pearson correlation coefficient (PCC) were used to measure the relevance between the native values and predicted ones. Here, the value of a protein dihedral angle is in the range of [<inline-formula id="inf21">
<mml:math id="m31">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>18</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x25e6;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf22">
<mml:math id="m32">
<mml:mrow>
<mml:mn>18</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x25e6;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>]. Before evaluating the protein dihedral angle, the difference between the predicted value <inline-formula id="inf23">
<mml:math id="m33">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and the actual value <italic>E</italic> is usually first converted to the dihedral angle according to <xref ref-type="disp-formula" rid="e11">Equation 11</xref> (<xref ref-type="bibr" rid="B19">Li et al., 2017</xref>). Then, the PCC and MAE are calculated by <xref ref-type="disp-formula" rid="e12">Equations 12</xref>, <xref ref-type="disp-formula" rid="e13">13,</xref> respectively.<disp-formula id="e11">
<mml:math id="m34">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mspace width="2.77695pt" class="tmspace"/>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>18</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x25e6;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>36</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x25e6;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mspace width="2.77695pt" class="tmspace"/>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>18</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x25e6;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>36</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x25e6;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mspace width="2.77695pt" class="tmspace"/>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>18</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x25e6;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf24">
<mml:math id="m35">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the original value of the predicted dihedral angle.<disp-formula id="e12">
<mml:math id="m36">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>Among them, <inline-formula id="inf25">
<mml:math id="m37">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf26">
<mml:math id="m38">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> are the means of <inline-formula id="inf27">
<mml:math id="m39">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>and <inline-formula id="inf28">
<mml:math id="m40">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>,</inline-formula> respectively, and <inline-formula id="inf29">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf30">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the standard deviations of <inline-formula id="inf31">
<mml:math id="m43">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf32">
<mml:math id="m44">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, respectively.<disp-formula id="e13">
<mml:math id="m45">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-3">
<title>3.3 Evaluation on independent test datasets</title>
<p>In order to better evaluate the performance of our model, we compare the DCMA model with other representative methods on public independent CASP sets. We compared the performance of DCMA with SPOT-1D, Netsurfp-2.0, DeepRIN, CRNN2, etc., on the CASP12 and CASP13 datasets, as shown in <xref ref-type="table" rid="T1">Table 1</xref>. Values in brackets are PCC, and &#x201c;-&#x201d; denotes data that cannot be obtained publicly. We re-implemented the model of Netsurfp-2.0 and used the same datasets and features as DCMA. DCMA achieved (19.42, 28.72) of <inline-formula id="inf33">
<mml:math id="m46">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> on CASP12 and (19.2, 27.98) of <inline-formula id="inf34">
<mml:math id="m47">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> on CASP13 respectively. The DCMA performance is weaker than that of SPOT-1D and OPUS-TASS and better than Netsurfp-2.0, DeepRIN, RaptorX-Angle, and SPIDER3.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Comparison of prediction performance on the CASP12 and CASP13 datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Method</th>
<th colspan="2" align="center">
<inline-formula id="inf35">
<mml:math id="m48">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th colspan="2" align="center">
<inline-formula id="inf36">
<mml:math id="m49">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
</tr>
<tr>
<th align="center">CASP12</th>
<th align="center">CASP13</th>
<th align="center">CASP12</th>
<th align="center">CASP13</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SPIDER3</td>
<td align="center">21.12 (0.809)</td>
<td align="center">-</td>
<td align="center">35.67 (0.783)</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">RaptorX-Angle</td>
<td align="center">20.69 (0.788)</td>
<td align="center">-</td>
<td align="center">31.6 (0.813)</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">DeepRIN</td>
<td align="center">20.21 (0.838)</td>
<td align="center">-</td>
<td align="center">31.39 (0.834)</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">NetSurfP-2.0</td>
<td align="center">20.0 (&#x2212;)</td>
<td align="center">-</td>
<td align="center">31.2 (&#x2212;)</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">NetSurfP-<inline-formula id="inf37">
<mml:math id="m50">
<mml:mrow>
<mml:mn>2.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</td>
<td align="center">20.0 (0.832)</td>
<td align="center">20.16 (0.838)</td>
<td align="center">30.52 (0.843)</td>
<td align="center">29.7 (0.847)</td>
</tr>
<tr>
<td align="center">SPOT-1<inline-formula id="inf38">
<mml:math id="m51">
<mml:mrow>
<mml:mtext>D</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>
<xref ref-type="table-fn" rid="Tfn2">
<sup>b</sup>
</xref>
</td>
<td align="center">18.91 (<bold>0.84</bold>)</td>
<td align="center">18.7 (0.839)</td>
<td align="center">27.46 (<bold>0.866</bold>)</td>
<td align="center">26.64 (<bold>0.867</bold>)</td>
</tr>
<tr>
<td align="center">CRRNN2</td>
<td align="center">19.14 (0.842)</td>
<td align="center">-</td>
<td align="center">28.9 (0.853)</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">OPUS-<inline-formula id="inf39">
<mml:math id="m52">
<mml:mrow>
<mml:mtext>TASS</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>
<xref ref-type="table-fn" rid="Tfn3">
<sup>c</sup>
</xref>
</td>
<td align="center">
<bold>18.12</bold> (&#x2212;)</td>
<td align="center">
<bold>17.94</bold> (&#x2212;)</td>
<td align="center">
<bold>26.0</bold> (&#x2212;)</td>
<td align="center">
<bold>25.95</bold> (&#x2212;)</td>
</tr>
<tr>
<td align="center">DCMA</td>
<td align="center">19.42 (0.837)</td>
<td align="center">19.2 (<bold>0.846</bold>)</td>
<td align="center">28.72 (0.857)</td>
<td align="center">27.98 (0.859)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>Results are generated by our reproduced experiment.</p>
</fn>
<fn id="Tfn2">
<label>
<sup>b</sup>
</label>
<p>Results are from SPOT-1D&#x2019;s online service.</p>
</fn>
<fn id="Tfn3">
<label>
<sup>c</sup>
</label>
<p>Data are computed by their public predicted results.</p>
</fn>
<fn>
<p>Boldface numbers indicate the best performance, and &#x201c;-&#x201d; denotes data that cannot be obtained publicly.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The comparison between state-of-the-art models, including SPOT-1D, OPUS-TASS, OPUS-TASS2, and DCMA, is further analyzed on free modeling targets in <xref ref-type="table" rid="T2">Table 2</xref>, <xref ref-type="table" rid="T3">3</xref>. The prediction results of OPUS-TASS2 based on sequence features are compared. The sequence features of OPUS-TASS2 are 20-D PSSM, 30-D HMM, 7-D physicochemical properties, and 19-D PSP19 features, which are denoted as &#x201c;OPUS-TASS2 (76D).&#x201d; Predicting results on the CASP12-FM dataset show that the performance of SPOT-1D and OPUS-TASS is similar and better than others. The performance of OPUS-TASS2 is best on the CASP-FM (56) dataset. DCMA achieved better prediction performance on the more difficult CASP13-FM dataset. On the most difficult CASP14-FM dataset, the prediction performance of OPUS-TASS2 and DCMA is comparable and better than other predictors.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Results of predicting <inline-formula id="inf40">
<mml:math id="m53">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> by different predictors on free modeling targets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Predictor</th>
<th align="center">CASP-FM(56)</th>
<th align="center">CASP12-FM</th>
<th align="center">CASP13-FM</th>
<th align="center">CASP14-FM</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">NetSurfP-<inline-formula id="inf41">
<mml:math id="m54">
<mml:mrow>
<mml:mn>2.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
<xref ref-type="table-fn" rid="Tfn4">
<sup>a</sup>
</xref>
</td>
<td align="center">20.55 (0.835)</td>
<td align="center">22.66 (0.826)</td>
<td align="center">22.68 (0.8)</td>
<td align="center">22.32 (0.789)</td>
</tr>
<tr>
<td align="center">SPOT-1D-Single</td>
<td align="center">-</td>
<td align="center">25.43 (&#x2212;)</td>
<td align="center">25.13 (&#x2212;)</td>
<td align="center">-</td>
</tr>
<tr>
<td align="center">SPOT-1D</td>
<td align="center">19.39 (&#x2212;)</td>
<td align="center">22.22 (0.827)<xref ref-type="table-fn" rid="Tfn5">
<sup>b</sup>
</xref>
</td>
<td align="center">
<bold>20.8</bold> (0.808)<xref ref-type="table-fn" rid="Tfn5">
<sup>b</sup>
</xref>
</td>
<td align="center">23.19 (&#x2212;)</td>
</tr>
<tr>
<td align="center">OPUS-TASS</td>
<td align="center">18.85 (&#x2212;)</td>
<td align="center">
<bold>21.9</bold> (&#x2212;)<xref ref-type="table-fn" rid="Tfn6">
<sup>c</sup>
</xref>
</td>
<td align="center">22.15 (&#x2212;)<xref ref-type="table-fn" rid="Tfn6">
<sup>c</sup>
</xref>
</td>
<td align="center">21.91 (&#x2212;)</td>
</tr>
<tr>
<td align="center">OPUS-TASS2 (76D)</td>
<td align="center">
<bold>18.58 (&#x2212;)</bold>
</td>
<td align="center">-</td>
<td align="center">-</td>
<td align="center">
<bold>21.79(&#x2212;)</bold>
</td>
</tr>
<tr>
<td align="center">DCMA</td>
<td align="center">19.44 (0.847)</td>
<td align="center">22.19 (0.831)</td>
<td align="center">20.84 (0.817)</td>
<td align="center">21.8 (0.794)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn4">
<label>
<sup>a</sup>
</label>
<p>Results are generated by our reproduced experiment.</p>
</fn>
<fn id="Tfn5">
<label>
<sup>b</sup>
</label>
<p>Results are from SPOT-1D&#x2019;s online service.</p>
</fn>
<fn id="Tfn6">
<label>
<sup>c</sup>
</label>
<p>The results are obtained locally using the OPUS-TASS standalone package.</p>
</fn>
<fn>
<p>Boldface numbers indicate the best performance, and &#x201c;-&#x201d; denotes data that cannot be obtained publicly.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Results of predicting <inline-formula id="inf42">
<mml:math id="m55">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> by different predictors on free modeling targets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Predictor</th>
<th align="center">CASP-FM(56)</th>
<th align="center">CASP12-FM</th>
<th align="center">CASP13-FM</th>
<th align="center">CASP14-FM</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">NetSurfP-2.0<xref ref-type="table-fn" rid="Tfn7">
<sup>a</sup>
</xref>
</td>
<td align="center">32.1 (0.832)</td>
<td align="center">36.3 (0.816)</td>
<td align="center">34.55 (0.821)</td>
<td align="center">41.1 (0.759)</td>
</tr>
<tr>
<td align="center">SPOT-1D-Single</td>
<td align="center">&#x2013;</td>
<td align="center">43.46 (&#x2212;)</td>
<td align="center">45.23 (&#x2212;)</td>
<td align="center">&#x2013;</td>
</tr>
<tr>
<td align="center">SPOT-1D</td>
<td align="center">30.1 (&#x2212;)</td>
<td align="center">34.71 (0.83)<xref ref-type="table-fn" rid="Tfn8">
<sup>b</sup>
</xref>
</td>
<td align="center">29.96 (0.853)<xref ref-type="table-fn" rid="Tfn8">
<sup>b</sup>
</xref>
</td>
<td align="center">43.98 (&#x2212;)</td>
</tr>
<tr>
<td align="center">OPUS-TASS</td>
<td align="center">28 (&#x2212;)</td>
<td align="center">
<bold>33.63(&#x2212;)</bold>
<xref ref-type="table-fn" rid="Tfn9">
<sup>c</sup>
</xref>
</td>
<td align="center">32.4 (&#x2212;)<xref ref-type="table-fn" rid="Tfn9">
<sup>c</sup>
</xref>
</td>
<td align="center">38.93 (&#x2212;)</td>
</tr>
<tr>
<td align="center">OPUS-TASS2 (76D)</td>
<td align="center">
<bold>26.91</bold>
</td>
<td align="center">&#x2013;</td>
<td align="center">&#x2013;</td>
<td align="center">38.65</td>
</tr>
<tr>
<td align="center">DCMA</td>
<td align="center">29.31 (0.854)</td>
<td align="center">34.66 (0.833)</td>
<td align="center">
<bold>29.9(0.854)</bold>
</td>
<td align="center">
<bold>38.57(0.772)</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn7">
<label>
<sup>a</sup>
</label>
<p>Results are generated by our reproduced experiment.</p>
</fn>
<fn id="Tfn8">
<label>
<sup>b</sup>
</label>
<p>Results are from SPOT-1D&#x2019;s online service.</p>
</fn>
<fn id="Tfn9">
<label>
<sup>c</sup>
</label>
<p>The results are obtained locally using the OPUS-TASS standalone package.</p>
</fn>
<fn>
<p>Boldface numbers indicate the best performance, and &#x201c;-&#x201d; denotes data that cannot be obtained publicly.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The latest benchmark dataset CASP15 (removing four similar sequences) was also used for evaluating the model generalization and achieved (19.78, 29.53) of MAE metrics and (0.83, 0.847) of PCC metrics on <inline-formula id="inf43">
<mml:math id="m56">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. For a more detailed understanding of the predicted results, the average dihedral angle prediction errors (measured by the MAE) are demonstrated on the CASP-FM(56), CASP12-FM, CASP13-FM, and CASP14-FM datasets across eight types of secondary structures, as shown in <xref ref-type="table" rid="T4">Table 4</xref>. The prediction errors of both H and E are the lowest because the secondary structures of H and E have the most samples.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>MAE of torsion angles on the 8-class secondary structures for the free modeling targets of the CASP datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="left"/>
<th align="center">H</th>
<th align="center">B</th>
<th align="center">E</th>
<th align="center">G</th>
<th align="center">I</th>
<th align="center">T</th>
<th align="center">S</th>
<th align="center">C</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="center">CASP-FM(56)</td>
<td align="left">phi</td>
<td align="right">11.42</td>
<td align="right">45.52</td>
<td align="right">25.38</td>
<td align="right">29.27</td>
<td align="right">0</td>
<td align="right">38.81</td>
<td align="right">62.66</td>
<td align="right">45.4</td>
</tr>
<tr>
<td align="left">psi</td>
<td align="right">7.39</td>
<td align="right">27.76</td>
<td align="right">19.63</td>
<td align="right">16.29</td>
<td align="right">0</td>
<td align="right">28.89</td>
<td align="right">34.78</td>
<td align="right">29.41</td>
</tr>
<tr>
<td rowspan="2" align="center">CASP12-FM</td>
<td align="left">phi</td>
<td align="right">8.78</td>
<td align="right">27.31</td>
<td align="right">20.97</td>
<td align="right">16.87</td>
<td align="right">0</td>
<td align="right">29.87</td>
<td align="right">38.16</td>
<td align="right">31.68</td>
</tr>
<tr>
<td align="left">psi</td>
<td align="right">15.03</td>
<td align="right">47.53</td>
<td align="right">28.74</td>
<td align="right">35.95</td>
<td align="right">0</td>
<td align="right">41.48</td>
<td align="right">61.76</td>
<td align="right">51.04</td>
</tr>
<tr>
<td rowspan="2" align="center">CASP13-FM</td>
<td align="left">phi</td>
<td align="right">7.34</td>
<td align="right">25.48</td>
<td align="right">20.38</td>
<td align="right">13.13</td>
<td align="right">0</td>
<td align="right">24.75</td>
<td align="right">40.80</td>
<td align="right">31.62</td>
</tr>
<tr>
<td align="left">psi</td>
<td align="right">10.70</td>
<td align="right">41.58</td>
<td align="right">21.17</td>
<td align="right">30.30</td>
<td align="right">0</td>
<td align="right">32.85</td>
<td align="right">69.57</td>
<td align="right">46.98</td>
</tr>
<tr>
<td rowspan="2" align="center">CASP14-FM</td>
<td align="left">phi</td>
<td align="right">8.44</td>
<td align="right">28.98</td>
<td align="right">22.57</td>
<td align="right">21.40</td>
<td align="right">0</td>
<td align="right">36.13</td>
<td align="right">36.72</td>
<td align="right">29.81</td>
</tr>
<tr>
<td align="left">psi</td>
<td align="right">16.53</td>
<td align="right">38.29</td>
<td align="right">32.74</td>
<td align="right">55.29</td>
<td align="right">0</td>
<td align="right">55.67</td>
<td align="right">69.81</td>
<td align="right">56.21</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>For the eight class definitions: G &#x3d; 3&#x2013;10 helix, H &#x3d; <inline-formula id="inf44">
<mml:math id="m57">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> helix, I &#x3d; <inline-formula id="inf45">
<mml:math id="m58">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> helix, B &#x3d; <inline-formula id="inf46">
<mml:math id="m59">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> bridge, E &#x3d; extended strand, S &#x3d; bend, T &#x3d; h-bonded turn, and C &#x3d; coil.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The prediction performance at the sequence level is further analyzed. The absolute error on the <inline-formula id="inf47">
<mml:math id="m60">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> angle of sequence T1039-D1, selected from the CASP14-FM dataset, is visualized in <xref ref-type="fig" rid="F5">Figure 5</xref>. Although the MAE at the sequence level is 16.92, the absolute errors on some residues are large. The predictive performance is acceptable on continuous secondary structure regions. When the structural state changes, prediction errors are high. We suppose that the discontinuous regions cannot supply more contextual features.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Predicting the absolute error of the <inline-formula id="inf48">
<mml:math id="m61">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> angle on sequence T1039-D1. The secondary structure is linearized for visualization.</p>
</caption>
<graphic xlink:href="fbinf-04-1477909-g005.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>3.4 Ablation study</title>
<p>The impact of different groups of input features is first analyzed. The loss variation on the validation dataset is compared when the model was trained by using different combinations of input features. The experimental results are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. Compared to the models trained by the input of a single feature, the features of pairwise combinations are more efficient in reducing losses. When three groups of features, PSSM profile, HHM profile, and residue coding, are combined, the effect of reducing loss is best.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Model loss variation in the validation dataset. The comparison of the dihedral angle prediction performance of the iterative procedure using different input features on the validation dataset.</p>
</caption>
<graphic xlink:href="fbinf-04-1477909-g004.tif"/>
</fig>
<p>We further analyzed the DCMA model structure. The results of the ablation experiment are shown in <xref ref-type="table" rid="T5">Table 5</xref>, where values in the cell and the bracket are the MAEs of (<inline-formula id="inf49">
<mml:math id="m62">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf50">
<mml:math id="m63">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>). Assuming the same other hyper-parameters, the performance of four, five, and six stacked I2A1 modules is compared. The experimental results show that the model with five stacked I2A1 modules is more effective and has fewer parameters. The influence of the multi-head attention network is also analyzed. The DCMA model removed the multi-head attention block, and the effectiveness was weakened.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Comparison of different stacked I2A1 blocks with the same hyper-parameters.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Test dataset</th>
<th align="center">Four blocks</th>
<th align="center">Five blocks</th>
<th align="center">Six blocks</th>
<th align="center">Five blocks without attention</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">CASP12</td>
<td align="center">19.57 (29.27)</td>
<td align="center">19.42 (<bold>28.72</bold>)</td>
<td align="center">
<bold>19.32</bold> (28.82)</td>
<td align="center">19.44 (28.99)</td>
</tr>
<tr>
<td align="center">CASP13</td>
<td align="center">18.91 (28.63)</td>
<td align="center">
<bold>19.2(27.98)</bold>
</td>
<td align="center">19.35 (28.31)</td>
<td align="center">19.32 (28.23)</td>
</tr>
<tr>
<td align="center">CASP12-FM</td>
<td align="center">22.4 (35.56)</td>
<td align="center">22.19 (<bold>34.66</bold>)</td>
<td align="center">
<bold>22.05</bold> (35.16)</td>
<td align="center">22.35 (35.19)</td>
</tr>
<tr>
<td align="center">CASP13-FM</td>
<td align="center">21.28 (31.42)</td>
<td align="center">
<bold>20.84(29.9)</bold>
</td>
<td align="center">22.13 (31.01)</td>
<td align="center">22.37 (31.70)</td>
</tr>
<tr>
<td align="center">CASP14-FM</td>
<td align="center">21.91 (39.78)</td>
<td align="center">
<bold>21.8(38.57)</bold>
</td>
<td align="center">22.16 (39.83)</td>
<td align="center">22.52 (39.43)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Boldface numbers indicate the best performance.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The length distributions of training and validation datasets are shown in <xref ref-type="fig" rid="F6">Figure 6</xref>. The data statistics show a total of 2022 sequences ranging in length from 102 to 213. In addition, there are 7,326 sequences with lengths ranging from 50 to 300. The recommended length of input sequence ranges from 50 to 300.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Length distributions of training and validation datasets.</p>
</caption>
<graphic xlink:href="fbinf-04-1477909-g006.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>Predicting protein 3D structures is an important and challenging task. Predicting protein backbone torsion dihedral angles helps solve the problem. Heavy models are unfriendly and unsuitable for running on edge computing devices. In particular, the file size of the SPOT-1D model is larger than 10 GB. In this article, a lightweight, faster, and individual model named DCMA is proposed. The model file of DCMA is less than 50 MB. We use hybrid dilated CNN and multi-head attention to design a new deep network structure, I2A1, that substitutes for RNN. The I2A1 block balanced the model generalization and computational efficiency well. Thus, our model mechanism can be applied to predicting various other protein attributes as well.</p>
<p>In future work, input residues will be characterized with more structural information, including physicochemical properties and protein domains (<xref ref-type="bibr" rid="B9">Guo et al., 2003</xref>; <xref ref-type="bibr" rid="B42">Yu et al., 2023</xref>), to improve the performance of discontinuous or isolated secondary structures. Although DCMA is more lightweight and faster, its input still relies on multi-sequence alignment information such as a PSSM profile. A single-sequence-based method that did not use evolutionary features would be more friendly.</p>
<p>Pre-trained protein language models (pLMs) can generate information-rich representations of sequences. Combined with sequence embedding generated by pLMs, a downstream predictor of backbone dihedral angles and other 1D structural properties can be exploited without generating multi-sequence alignment information.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material; further inquiries can be directed to the corresponding author/s.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>BZ: conceptualization, methodology, software, and writing&#x2013;original draft. MZ: conceptualization, methodology, and writing&#x2013;original draft. YZ: conceptualization, data curation, and writing&#x2013;review and editing. LQ: conceptualization, data curation, and writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research was supported in part by the Excellent Youth Scholars Project of Anhui Provincial Universities of China under grant no. gxyq2020029 and the Project of Provincial Key Laboratory for Computer Information Processing Technology, Soochow University of China, under grant no. KJS1934.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altschul</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Madden</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Schffer</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Webb</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>1997</year>). <article-title>Gapped blast and psi-blast: a new generation of protein database search programs</article-title>. <source>Nucleic acids Res.</source> <volume>25</volume>, <fpage>3389</fpage>&#x2013;<lpage>3402</lpage>. <pub-id pub-id-type="doi">10.1093/nar/25.17.3389</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Papandreou</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Schroff</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Adam</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Rethinking atrous convolution for semantic image segmentation</source>. <comment>CoRR abs/1706.05587</comment>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheung</surname>
<given-names>M.-S.</given-names>
</name>
<name>
<surname>Maguire</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Stevens</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Broadhurst</surname>
<given-names>R. W.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Dangle: a bayesian inferential method for predicting protein backbone dihedral angles and secondary structure</article-title>. <source>J. magnetic Reson.</source> <volume>202</volume>, <fpage>223</fpage>&#x2013;<lpage>233</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmr.2009.11.008</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Du</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>The trrosetta server for fast and accurate protein structure prediction</article-title>. <source>Nat. Protoc.</source> <volume>16</volume>, <fpage>5634</fpage>&#x2013;<lpage>5651</lpage>. <pub-id pub-id-type="doi">10.1038/s41596-021-00628-9</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2018a</year>). <article-title>Mufold-ss:new deep inception-inside-inception networks for protein secondary structure prediction</article-title>. <source>Proteins Struct. Funct. Bioinforma.</source> <volume>86</volume>, <fpage>592</fpage>&#x2013;<lpage>598</lpage>. <pub-id pub-id-type="doi">10.1002/prot.25487</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2018b</year>). <article-title>Prediction of protein backbone torsion angles using deep residual inception neural networks</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>16</volume>, <fpage>1020</fpage>&#x2013;<lpage>1028</lpage>. <pub-id pub-id-type="doi">10.1109/tcbb.2018.2814586</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Raptorx-angle: real-value prediction of protein backbone dihedral angles through a hybrid method of clustering and deep learning</article-title>. <source>BMC Bioinforma.</source> <volume>19</volume>, <fpage>100</fpage>&#x2013;<lpage>184</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2065-x</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Graves</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jaitly</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Mohamed</surname>
<given-names>A. R.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Hybrid speech recognition with deep bidirectional lstm</article-title>,&#x201d; in <source>Automatic speech recognition and understanding</source>, <fpage>273</fpage>&#x2013;<lpage>278</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Improving the performance of domainparser for structural domain partition using neural network</article-title>. <source>Nucleic Acids Res.</source> <volume>31</volume>, <fpage>944</fpage>&#x2013;<lpage>952</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkg189</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hanson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Paliwal</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Litfin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Improving prediction of protein secondary structure, backbone angles, solvent accessibility and contact numbers by using predicted contact maps and an ensemble of recurrent and residual convolutional neural networks</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>2403</fpage>&#x2013;<lpage>2410</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty1006</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Deep residual learning for image recognition</article-title>,&#x201d; in <source>2016 IEEE conference on computer vision and pattern recognition (CVPR)</source> (<publisher-loc>Los Alamitos, CA, USA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>770</fpage>&#x2013;<lpage>778</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heffernan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Paliwal</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Capturing non-local interactions by long short-term memory bidirectional recurrent neural networks for improving prediction of protein secondary structure, backbone angles, contact numbers and solvent accessibility</article-title>. <source>Bioinformatics</source> <volume>33</volume>, <fpage>2842</fpage>&#x2013;<lpage>2849</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx218</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hochreiter</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Schmidhuber</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Long short-term memory</article-title>. <source>Neural Comput.</source> <volume>9</volume>, <fpage>1735</fpage>&#x2013;<lpage>1780</lpage>. <pub-id pub-id-type="doi">10.1162/neco.1997.9.8.1735</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ismi</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Pulungan</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Afiahayati</surname>
</name>
</person-group> (<year>2022</year>). <article-title>Deep learning for protein secondary structure prediction: pre and post-alphafold</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>20</volume>, <fpage>6271</fpage>&#x2013;<lpage>6286</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2022.11.012</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jozefowicz</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zaremba</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>An empirical exploration of recurrent network architectures</article-title>,&#x201d; in <source>Proceedings of the 32nd international converenfe on machine learning (ICML)</source>, <fpage>171</fpage>&#x2013;<lpage>180</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jumper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Figurnov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ronneberger</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Highly accurate protein structure prediction with alphafold</article-title>. <source>Nature</source> <volume>596</volume>, <fpage>583</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname>
<given-names>H. S.</given-names>
</name>
<name>
<surname>Kurochkina</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>Estimation and use of protein backbone angle probabilities</article-title>. <source>J. Mol. Biol.</source> <volume>229</volume>, <fpage>448</fpage>&#x2013;<lpage>460</lpage>. <pub-id pub-id-type="doi">10.1006/jmbi.1993.1045</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klausen</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Jespersen</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Nielsen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jensen</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Jurtz</surname>
<given-names>V. I.</given-names>
</name>
<name>
<surname>Soenderby</surname>
<given-names>C. K.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Netsurfp-2.0: improved prediction of protein structural features by integrated deep learning</article-title>. <source>Proteins Struct. Funct. Bioinforma.</source> <volume>87</volume>, <fpage>520</fpage>&#x2013;<lpage>527</lpage>. <pub-id pub-id-type="doi">10.1002/prot.25674</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Adhikari</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Lyu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Deep learning methods for protein torsion angle prediction</article-title>. <source>BMC Bioinforma.</source> <volume>18</volume>, <fpage>417</fpage>&#x2013;<lpage>513</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-017-1834-2</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schuster</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Paliwal</surname>
<given-names>K. K.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Bidirectional recurrent neural networks</article-title>. <source>IEEE Trans. Signal Process.</source> <volume>45</volume>, <fpage>2673</fpage>&#x2013;<lpage>2681</lpage>. <pub-id pub-id-type="doi">10.1109/78.650093</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Delaglio</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cornilescu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bax</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Talos&#x2b;: a hybrid method for predicting protein backbone torsion angles from nmr chemical shifts</article-title>. <source>J. Biomol. NMR</source> <volume>44</volume>, <fpage>213</fpage>&#x2013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1007/s10858-009-9333-z</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Raghava</surname>
<given-names>G. P. S.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Evaluation of protein dihedral angle prediction methods</article-title>. <source>PLOS ONE</source> <volume>9</volume>, <fpage>e105667</fpage>&#x2013;<lpage>e105669</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0105667</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Litfin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Paliwal</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hanumanthappa</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021a</year>). <article-title>Spot-1d-single: improving the single-sequence-based prediction of protein secondary structure, backbone angles, solvent accessibility and half-sphere exposures using a large training set and ensembled deep learning</article-title>. <source>Bioinformatics</source> <volume>37</volume>, <fpage>3464</fpage>&#x2013;<lpage>3472</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab316</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Paliwal</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021b</year>). <article-title>Rna backbone torsion and pseudotorsion angle prediction using dilated convolutional neural networks</article-title>. <source>J. Chem. Inf. Model.</source> <volume>61</volume>, <fpage>2610</fpage>&#x2013;<lpage>2622</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.1c00153</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Webb</surname>
<given-names>G. I.</given-names>
</name>
<name>
<surname>Akutsu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Tangle: two-level support vector regression approach for protein backbone torsion angle prediction from primary sequences</article-title>. <source>PloS one</source> <volume>7</volume>, <fpage>e30361</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0030361</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Steinegger</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Meier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mirdita</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>V&#xf6;hringer</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>S&#xf6;ding</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>S&#xf6;ding</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>HH-suite3 for fast remote homology detection and deep protein annotation</article-title>. <source>BMC Bioinforma.</source> <volume>20</volume>, <fpage>473</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-019-3019-7</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Strubell</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Verga</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Belanger</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mccallum</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Fast and accurate entity recognition with iterated dilated convolutions</article-title>,&#x201d; in <source>Proceedings of the 2017 conference on empirical methods in natural language processing (EMNLP)</source>, <fpage>2670</fpage>&#x2013;<lpage>2680</lpage>. <pub-id pub-id-type="doi">10.18653/v1/D17-1283</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sunghoon</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Se-Eun</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hyeon</surname>
<given-names>S. S.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Validity of protein structure alignment method based on backbone torsion angles</article-title>. <source>J. Proteomics Bioinforma.</source> <volume>4</volume>, <fpage>218</fpage>&#x2013;<lpage>226</lpage>. <pub-id pub-id-type="doi">10.4172/jpb.1000190</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Szegedy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sermanet</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Reed</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Anguelov</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). &#x201c;<article-title>Going deeper with convolutions</article-title>,&#x201d; in <source>2015 IEEE conference on computer vision and pattern recognition (CVPR)</source>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2015.7298594</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Szegedy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Vanhoucke</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Ioffe</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shlens</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wojna</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Rethinking the inception architecture for computer vision</article-title>,&#x201d; in <source>2016 IEEE conference on computer vision and pattern recognition (CVPR)</source> (<publisher-name>IEEE Computer Society</publisher-name>), <fpage>2818</fpage>&#x2013;<lpage>2826</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tay</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dehghani</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bahri</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Metzler</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Efficient transformers: a survey</source>. <comment>CoRR abs/2009.06732</comment>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uddin</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Mahbub</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rahman</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Bayzid</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>SAINT: self-attention augmented inception-inside-inception network improves protein secondary structure prediction</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>4599</fpage>&#x2013;<lpage>4608</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa531</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <source>Attention is all you need</source>. <comment>CoRR abs/1706.03762</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.1706.03762</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dunbrack</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Roland</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Pisces: a protein sequence culling server</article-title>. <source>Bioinformatics</source> <volume>19</volume>, <fpage>1589</fpage>&#x2013;<lpage>1591</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btg224</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Understanding convolution for semantic segmentation</article-title>,&#x201d; in <source>2018 IEEE winter conference on applications of computer vision (WACV)</source>, <fpage>1451</fpage>&#x2013;<lpage>1460</lpage>. <pub-id pub-id-type="doi">10.1109/WACV.2018.00163</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wood</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Hirst</surname>
<given-names>J. D.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Protein secondary structure prediction with dihedral angles</article-title>. <source>PROTEINS Struct. Funct. Bioinforma.</source> <volume>59</volume>, <fpage>476</fpage>&#x2013;<lpage>481</lpage>. <pub-id pub-id-type="doi">10.1002/prot.20435</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Anglor: a composite machine-learning algorithm for protein backbone torsion angle prediction</article-title>. <source>PloS one</source> <volume>3</volume>, <fpage>e3400</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0003400</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>van den Hengel</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Bridging category-level and instance-level semantic image segmentation</source>. <comment>CoRR abs/1605.06885</comment>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Opus-tass: a protein backbone torsion angles and secondary structure predictor based on ensemble neural networks</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>5021</fpage>&#x2013;<lpage>5026</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa629</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>OPUS-X: an open-source toolkit for protein torsion angles, secondary structure, solvent accessibility, contact map predictions and 3D folding</article-title>. <source>Bioinformatics</source> <volume>38</volume>, <fpage>108</fpage>&#x2013;<lpage>114</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab633</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Koltun</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Multi-scale context aggregation by dilated convolutions</article-title>,&#x201d; in <conf-name>Conference Track Proceedings 4th International Conference on Learning Representations, ICLR 2016</conf-name>, <conf-loc>San Juan, Puerto Rico</conf-loc>, <conf-date>May 2-4, 2016</conf-date>. Editors <person-group person-group-type="editor">
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
</person-group>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Z.-Z.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>C.-X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.-G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>G.-J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Dombpred: protein domain boundary prediction based on domain-residue clustering using inter-residue distance</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinforma.</source> <volume>20</volume>, <fpage>912</fpage>&#x2013;<lpage>922</lpage>. <pub-id pub-id-type="doi">10.1109/TCBB.2022.3175905</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>L&#xfc;</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Prediction of 8-state protein secondary structures by a novel deep learning architecture</article-title>. <source>BMC Bioinforma.</source> <volume>19</volume>, <fpage>293</fpage>&#x2013;<lpage>313</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2280-5</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Quan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lyu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Multi-task deep learning for concurrent prediction of protein structural properties</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/2021.02.04.429840</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Accurate prediction of protein dihedral angles through conditional random field</article-title>. <source>Front. Biol.</source> <volume>8</volume>, <fpage>353</fpage>&#x2013;<lpage>361</lpage>. <pub-id pub-id-type="doi">10.1007/s11515-013-1261-3</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Litfin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>3 &#x3d; 1 &#x2b; 2: how the divide conquered <italic>de novo</italic> protein structure prediction and what is next?</article-title> <source>Natl. Sci. Rev.</source> <volume>10</volume>, <fpage>nwad259</fpage>. <pub-id pub-id-type="doi">10.1093/nsr/nwad259</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>