<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mol. Biosci.</journal-id>
<journal-title>Frontiers in Molecular Biosciences</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mol. Biosci.</abbrev-journal-title>
<issn pub-type="epub">2296-889X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">863141</article-id>
<article-id pub-id-type="doi">10.3389/fmolb.2022.863141</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Molecular Biosciences</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Disordered&#x2013;Ordered Protein Binary Classification by Circular Dichroism Spectroscopy</article-title>
<alt-title alt-title-type="left-running-head">Micsonai et al.</alt-title>
<alt-title alt-title-type="right-running-head">Disordered&#x2013;Ordered Classification by CD Spectroscopy</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Micsonai</surname>
<given-names>Andr&#xe1;s</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1654084/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Moussong</surname>
<given-names>&#xc9;va</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1654067/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Murvai</surname>
<given-names>Nikoletta</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1732654/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tantos</surname>
<given-names>&#xc1;gnes</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1730533/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>T&#x151;ke</surname>
<given-names>Orsolya</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>R&#xe9;fr&#xe9;giers</surname>
<given-names>Matthieu</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/294836/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wien</surname>
<given-names>Frank</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kardos</surname>
<given-names>J&#xf3;zsef</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/989141/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>ELTE NAP Neuroimmunology Research Group</institution>, <institution>Department of Biochemistry</institution>, <institution>Institute of Biology</institution>, <institution>ELTE E&#xf6;tv&#xf6;s Lor&#xe1;nd University</institution>, <addr-line>Budapest</addr-line>, <country>Hungary</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Biochemistry</institution>, <institution>Institute of Biology</institution>, <institution>ELTE E&#xf6;tv&#xf6;s Lor&#xe1;nd University</institution>, <addr-line>Budapest</addr-line>, <country>Hungary</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Institute of Enzymology</institution>, <institution>Research Centre for Natural Sciences</institution>, <addr-line>Budapest</addr-line>, <country>Hungary</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Laboratory for NMR Spectroscopy</institution>, <institution>Research Centre for Natural Sciences</institution>, <addr-line>Budapest</addr-line>, <country>Hungary</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Synchrotron SOLEIL</institution>, <addr-line>Gif-sur-Yvette</addr-line>, <country>France</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Centre de Biophysique Mol&#xe9;culaire</institution>, <institution>CNRS UPR4301</institution>, <addr-line>Orl&#xe9;ans</addr-line>, <country>France</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/32277/overview">Vladimir N. Uversky</ext-link>, University of South Florida, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/561321/overview">Kundlik Gadhave</ext-link>, Johns Hopkins University, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/599844/overview">Nicolas Palopoli</ext-link>, National University of Quilmes, Argentina</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: J&#xf3;zsef Kardos, <email>kardos@elte.hu</email>
</corresp>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work</p>
</fn>
<fn fn-type="other">
<p>This article was submitted to Protein Folding, Misfolding and Degradation, a section of the journal Frontiers in Molecular Biosciences</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>9</volume>
<elocation-id>863141</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>03</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Micsonai, Moussong, Murvai, Tantos, T&#x151;ke, R&#xe9;fr&#xe9;giers, Wien and Kardos.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Micsonai, Moussong, Murvai, Tantos, T&#x151;ke, R&#xe9;fr&#xe9;giers, Wien and Kardos</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Intrinsically disordered proteins lack a stable tertiary structure and form dynamic conformational ensembles due to their characteristic physicochemical properties and amino acid composition. They are abundant in nature and responsible for a large variety of cellular functions. While numerous bioinformatics tools have been developed for <italic>in silico</italic> disorder prediction in the last decades, there is a need for experimental methods to verify the disordered state. CD spectroscopy is widely used for protein secondary structure analysis. It is usable in a wide concentration range under various buffer conditions. Even without providing high-resolution information, it is especially useful when NMR, X-ray, or other techniques are problematic or one simply needs a fast technique to verify the structure of proteins. Here, we propose an automatized binary disorder&#x2013;order classification method by analyzing far-UV CD spectroscopy data. The method needs CD data at only three wavelength points, making high-throughput data collection possible. The mathematical analysis applies the <italic>k</italic>-nearest neighbor algorithm with cosine distance function, which is independent of the spectral amplitude and thus free of concentration determination errors. Moreover, the method can be used even for strong absorbing samples, such as the case of crowded environmental conditions, if the spectrum can be recorded down to the wavelength of 212&#xa0;nm. We believe the classification method will be useful in identifying disorder and will also facilitate the growth of experimental data in IDP databases. The method is implemented on a webserver and freely available for academic users.</p>
</abstract>
<kwd-group>
<kwd>intrinsically disordered proteins</kwd>
<kwd>CD spectroscopy</kwd>
<kwd>protein secondary structure</kwd>
<kwd>disorder identifier</kwd>
<kwd>disorder&#x2013;order classification</kwd>
<kwd>machine learning</kwd>
</kwd-group>
<contract-num rid="cn001">2017-1.2.1-NKP-2017-00002 PD135510 K120391 K125340 K131702 K138937</contract-num>
<contract-sponsor id="cn001">National Research, Development and Innovation Office<named-content content-type="fundref-id">10.13039/501100018818</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Intrinsically disordered proteins (IDPs) or protein regions (IDRs) lack a stable tertiary structure and form dynamic conformational ensembles (<xref ref-type="bibr" rid="B6">Dunker et al., 2002</xref>; <xref ref-type="bibr" rid="B7">Habchi et al., 2014</xref>). They are abundant in nature, especially in eukaryotes, and responsible for a plethora of cellular functions (<xref ref-type="bibr" rid="B27">Peng et al., 2015</xref>). Overall, 3&#x2013;17% of eukaryotic proteins are estimated to be fully disordered (<xref ref-type="bibr" rid="B5">Dunker et al., 2000</xref>), and 30&#x2013;50% of proteins contain IDRs (<xref ref-type="bibr" rid="B5">Dunker et al., 2000</xref>; <xref ref-type="bibr" rid="B44">Ward et al., 2004</xref>). The recently published state-of-the-art structure prediction method, AlphaFold2, provides confident prediction for only 58% of the residues on nearly the entire human proteome (<xref ref-type="bibr" rid="B37">Tunyasuvunakool et al., 2021</xref>), indicating that more than 40% of the residues fall into regions with significant structural flexibility. The biological importance of disordered proteins is underlined by the fact that malfunction of IDPs can lead to a variety of diseases (<xref ref-type="bibr" rid="B41">Uversky et al., 2008</xref>; <xref ref-type="bibr" rid="B32">Ruan et al., 2019</xref>). Given that IDPs have fundamentally different physicochemical properties than globular proteins, identifying disordered proteins and regions based on the amino acid sequence is highly desirable. In the last decade, dozens of bioinformatics tools have been developed to predict intrinsic disorder and its molecular function (<xref ref-type="bibr" rid="B43">Varadi et al., 2015</xref>; <xref ref-type="bibr" rid="B13">Katuwawala et al., 2020</xref>). Although these tools provide fast and high-throughput analysis, they have a substantial error rate and the actual predictions need experimental verification. The main experimental techniques applied to investigate intrinsic disorder include NMR, X-ray, circular dichroism (CD) spectroscopy, cryo-EM, and other spectroscopic techniques and techniques that study the hydrodynamic radius or surface exposure. Despite significant efforts to characterize structural disorder in detail, our knowledge remains limited. Even DisProt, the largest database of manually curated, experimentally verified disordered proteins and regions (<xref ref-type="bibr" rid="B31">Quaglia et al., 2021</xref>), only contains annotations of around 2000 proteins covering a small fraction of the predicted amount. Most of the structure characterization methods have high time and sample requirements; hence, there is a high need for fast, high-throughput, and inexpensive experimental methods to verify disorder.</p>
<p>CD spectroscopy has been widely used to study the structure of proteins. Near-UV CD spectra in the 250&#x2013;300&#xa0;nm wavelength range are determined by the aromatic side chains and their environment. In disordered conformation, these side chains are accessible for the polar solvent, and their environment is averaged out resulting in a nearly zero CD signal. Therefore, such a low signal could be the sign of disorder; however, IDPs usually contain a low number of aromatic residues, which restricts the practical use of this method. Moreover, near-UV CD needs a relatively large amount of sample because of the long path length and high required protein concentration (<xref ref-type="bibr" rid="B46">Woody and Berova, 2000</xref>). Far-UV CD spectra are characteristic of the secondary structure of proteins and need two orders of magnitude less amount of protein for the measurements than near-UV measurements. Disordered proteins exhibit characteristic CD spectra with an intensive minimum in the vicinity of 200&#xa0;nm and a low amplitude around 222&#xa0;nm (<xref ref-type="bibr" rid="B1">Adler et al., 1973</xref>; <xref ref-type="bibr" rid="B29">Provencher and Gloeckner, 1981</xref>; <xref ref-type="bibr" rid="B9">Johnson, 1988</xref>; <xref ref-type="bibr" rid="B15">Kelly and Price, 1997</xref>; <xref ref-type="bibr" rid="B38">Uversky, 1999</xref>). Uversky and co-workers reported that by using these two wavelengths for a double plot, it is possible to distinguish intrinsically disordered proteins from the ones with high secondary structure contents, such as molten globules and native globular proteins (<xref ref-type="bibr" rid="B40">Uversky, 2002</xref>; <xref ref-type="bibr" rid="B42">Uversky, 2003</xref>; <xref ref-type="bibr" rid="B39">Uversky and Fink, 2004</xref>). Our aim in the present work was to revise this observation and work out an automatized method for improved identification of IDPs by CD spectroscopy. We collected a larger reference set of CD spectra of ordered globular proteins and disordered polypeptide chains based on our own measurements, data downloaded from the protein CD database (PCDDB) (<xref ref-type="bibr" rid="B45">Whitmore et al., 2017</xref>), and collected from the literature. Starting with the double-wavelength plot, we applied various algorithms searching for an optimal method to identify disordered proteins from the spectral information gathered by CD spectroscopy. We examined the number and values of wavelengths needed for accurate disorder detection. To find the optimal method, the robustness regarding the sensitivity for incorrect concentration determination and experimental noise were also taken into account. Based on our findings, we provide a thorough comparison of the various analysis methods and propose an optimal protocol for IDP detection.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methods</title>
<sec id="s2-1">
<title>CD Spectroscopy</title>
<p>Synchrotron radiation CD (SRCD) spectra were recorded at the DISCO beamline of SOLEIL French synchrotron facility (proposal Nos. 20181890, 20191810, and 20200751). Samples at 5&#x2013;7&#xa0;mg/ml were measured in CaF<sub>2</sub> cells with path lengths of 6&#x2013;20&#xa0;&#x3bc;m. In total, 6&#x2013;12 scans were accumulated in the 175&#x2013;270&#xa0;nm or 180&#x2013;270&#xa0;nm wavelength range depending on the sample absorption; 1&#xa0;nm data steps with a lock-in time constant of 300&#xa0;ms and integration time of 1,200&#xa0;ms were used. After baseline subtraction, the spectrum was corrected with the CSA calibration (<xref ref-type="bibr" rid="B4">Chen and Yang, 1977</xref>). Protein concentration was determined by directly measuring the absorbance of the CD sample and buffer reference at 205 and 214&#xa0;nm (<xref ref-type="bibr" rid="B17">Kuipers and Gruppen, 2007</xref>; <xref ref-type="bibr" rid="B2">Anthis and Clore, 2013</xref>). For the case studies, CD experiments were carried out on a Jasco J-810 spectropolarimeter (Japan Spectroscopic Co., Tokyo, Japan). Protein concentrations 10, 1, and 0.1&#xa0;mg/ml were used with quartz cells of 13&#xa0;&#x3bc;m, 103&#xa0;&#x3bc;m, and 1&#xa0;mm path lengths, respectively.</p>
</sec>
<sec id="s2-2">
<title>Mathematical Models Used for Disordered&#x2013;Ordered Binary Classification</title>
<p>For disordered&#x2013;ordered classification, the following built-in models of the MATLAB Classification Toolbox were used.</p>
<p>
<italic>Tree:</italic> A binary classification decision tree is a learning method, where internal nodes represent the inspection of a predictor, branches show the outcome of the inspection, and leaf nodes represent class labels. Based on the number of leaves, we categorized trees as &#x201c;simple&#x201d; and &#x201c;medium.&#x201d; The maximum number of leaves is 4 in a simple tree and 20 in a medium tree.</p>
<p>
<italic>Support vector machines:</italic> SVMs are methods which use a subset of training data to create a decision function. The data points in this subset are called support vectors. We used different kernel functions for our models: linear and radial basis function (RBF). SVM algorithms aim to find a hyperplane that separates two labeled classes with the widest possible margin.</p>
<p>
<italic>K-nearest neighbors:</italic> KNN classification is based on finding the k-nearest training point to the new data point and using them to predict the label. We used four types of KNN methods which calculate the Euclidean distance between data points. &#x201c;Fine,&#x201d; &#x201c;medium,&#x201d; and &#x201c;coarse&#x201d; examine 1, 10, and 100 nearest neighbors, respectively. &#x201c;Weighted&#x201d; applies a squared inverse distance weighting function on the 10 nearest neighbors, which results in nearer neighbors having a larger impact. The fifth KNN method we implemented uses a different distance metric; it considers the cosine of the angle between vectors pointing from the origin to data points searching for 10 nearest neighbors.</p>
<p>
<italic>Discriminant:</italic> Discriminant analyses create a decision surface, which may be linear or quadratic. In the case of &#x201c;diaglinear&#x201d; and &#x201c;diagquadratic&#x201d; models, the covariance matrices are diagonal (i.e., all the off-diagonal elements&#x2014;covariances&#x2014;are zeros; only variances are non-zero values). As opposed to SVM models, discriminant analyses do not include the condition of making margins as wide as possible.</p>
</sec>
<sec id="s2-3">
<title>Steps of Finding the Optimal Classification Method</title>
<p>We aimed to classify proteins based on two or three data points. Therefore, we implemented classifiers that consider either a pair or a triplet of wavelengths, and perform classification by using CD values at the given wavelengths. Wavelength pairs and triplets consisted of wavelengths with a minimum pairwise difference of 3&#xa0;nm in the 175&#x2013;250&#xa0;nm wavelength range. To develop the disorder determination method in a certain wavelength range, we used all proteins whose spectra covered the studied range.</p>
<p>Leave-one-out cross-validation error rates were calculated by summing misclassified proteins and dividing their number by the total number of proteins. Error rates were determined separately for disordered and ordered proteins and for the total dataset.</p>
<p>The robustness of each method was also tested. We simulated the effect of inaccurate concentration measurement by rescaling the amplitude of test spectra and examined the sensitivity of methods to the scaling factor in the range of 0.5&#x2013;2. Furthermore, the dependence of the methods&#x2019; accuracy on noise was evaluated. Noise was added independently to each CD value using random values from normal distribution (&#xb5; &#x3d; 0&#xa0;M<sup>&#x2212;1</sup> cm<sup>&#x2212;1</sup>, &#x3c3; &#x3d; 0.1&#xa0;M<sup>&#x2212;1</sup> cm<sup>&#x2212;1</sup> or &#x3c3; &#x3d; 0.05&#xa0;M<sup>&#x2212;1</sup> cm<sup>&#x2212;1</sup>). The effect of noise was calculated by averaging the results of 1,000 simulations.</p>
<p>When picking the best classifiers, the global error, error on disordered structure, preferably higher wavelengths for analysis, and the robustness were considered.</p>
<p>MATLAB scripts used in the present study are provided in the <xref ref-type="sec" rid="s10">Supplementary Material</xref>.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<title>Results and Discussion</title>
<sec id="s3-1">
<title>Reference Dataset of IDPs and Ordered Proteins</title>
<p>To investigate the problem of distinction between disordered and ordered protein structures based on CD data alone, we collected the CD spectra of IDPs and proteins with ordered structures from various sources. In total, 140 high-quality SRCD spectra in a wide wavelength range from 175 or 180&#xa0;nm of globular native proteins were downloaded from the protein CD databank (PCDDB) (<xref ref-type="bibr" rid="B45">Whitmore et al., 2017</xref>). The spectra of 9 globular native proteins, 2 amyloid fibrils, and 26 disordered polypeptides were the result of our SRCD measurements. These include IDPs, such as ERD14 (early responsive to dehydration) plant chaperone and its variants (<xref ref-type="bibr" rid="B23">Murvai et al., 2021</xref>), histone&#x2013;lysine N-methyltransferase constructs, artificial peptides designed for maximal disorder, and &#x3b2;-structure-rich globular proteins, such as dUTPase and SH3 domains that have CD spectra similar to disordered proteins. Overall, 85 spectra were collected from the literature (based on the references in <xref ref-type="bibr" rid="B40">Uversky (2002)</xref>, <xref ref-type="bibr" rid="B42">Uversky (2003)</xref>, <xref ref-type="bibr" rid="B39">Uversky and Fink (2004)</xref>), including those of 30 globular proteins and 55 IDPs. These spectra varied in their wavelength range. To develop the disorder prediction method in a certain wavelength range, we used all proteins whose spectra covered the studied range. The proteins of the reference set are presented in <xref ref-type="sec" rid="s10">Supplementary Table S1</xref>, and the size of the reference set as a function of the wavelength cutoff is presented in <xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>.</p>
</sec>
<sec id="s3-2">
<title>Classical CD Plot of IDPs and Ordered Proteins</title>
<p>We reproduced the double-wavelength plot using CD intensities at 200 and 222&#xa0;nm wavelengths on the available data on proteins reported by <xref ref-type="bibr" rid="B40">Uversky (2002)</xref>, <xref ref-type="bibr" rid="B42">Uversky (2003)</xref>, <xref ref-type="bibr" rid="B39">Uversky and Fink (2004)</xref>, as shown in <xref ref-type="fig" rid="F1">Figure 1A</xref>. IDPs and globular proteins were separated with some overlap in the plot. However, when we completed this plot with all the proteins in our database, this picture has changed significantly (<xref ref-type="fig" rid="F1">Figure 1B</xref>). Although the newly added spectra of disordered peptides concentrated well on the previous disordered ones, the globular proteins covered a much wider space and even overlapped with the disordered region ruining the spatial separation.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>2D-plot of CD data of IDPs and ordered proteins. <bold>(A)</bold> Mean residue ellipticities at 200 and 222&#xa0;nm wavelengths for IDPs (yellow) and globular proteins (light blue) were collected from the literature for proteins previously studied by <xref ref-type="bibr" rid="B40">Uversky (2002)</xref>, <xref ref-type="bibr" rid="B42">Uversky (2003)</xref>, <xref ref-type="bibr" rid="B39">Uversky and Fink (2004)</xref>. &#x201c;Random coil&#x201d; and &#x201c;premolten globule&#x201d; types of IDPs were not distinguished in our work. <bold>(B)</bold> Plot of the full reference database. IDPs over the ones presented in <bold>(A)</bold> are shown in red, while the additional globular ones are shown in dark blue. Hollow circles show those proteins that are incorrectly classified as disordered or ordered by using the 200 and 222&#xa0;nm wavelength data of proteins presented in panel A as training set for disordered&#x2013;ordered classification (see later). Note the large spectral (and conformational) space covered by the ordered proteins.</p>
</caption>
<graphic xlink:href="fmolb-09-863141-g001.tif"/>
</fig>
<p>The CD spectra of those globular proteins that are located in the disordered region in the double-wavelength plot are similar to that of the disordered ones, despite their fully ordered globular structure (<xref ref-type="fig" rid="F2">Figure 2</xref>). Their X-ray structures revealed that these proteins have highly right-hand twisted antiparallel &#x3b2;-sheet structures (<xref ref-type="bibr" rid="B8">Ho and Curmi, 2002</xref>; <xref ref-type="bibr" rid="B22">Micsonai et al., 2015</xref>) (<xref ref-type="fig" rid="F2">Figure 2</xref>). This problem has already been pointed out in our previous work (<xref ref-type="bibr" rid="B21">Micsonai et al., 2018</xref>) as a major issue in the distinction between highly twisted antiparallel &#x3b2;-sheets and disordered structures in secondary structure content estimation. These results reveal that the simple use of the 200 and 222&#xa0;nm CD data might be insufficient for the proper distinction between IDPs and ordered proteins, and an improvement of this methodology is highly beneficial.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>CD spectra of disordered proteins and some globular proteins with similar spectra. Proteins rich in highly twisted antiparallel &#x3b2;-sheets (colored spectra and corresponding structures) exhibit CD spectra reminiscent of disordered proteins (gray), which makes the distinction between them difficult. Alpha-chymotrypsin (PDB ID: 5CHA), chymotrypsinogen (2CGA), trypsin inhibitor (5PTI), elastase (3EST), ferredoxin (2FDN), ecotin (1ECZ), dUTP pyrophosphatase (1Q5U), and trypsin inhibitor (Kunitz) (1BA7) are shown.</p>
</caption>
<graphic xlink:href="fmolb-09-863141-g002.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>Identification of IDPs Using Various Mathematical Models</title>
<p>To develop a binary classification method (IDP vs. ordered structure) for an accurate and automatized IDP identification, we analyzed the CD spectra of our database using various mathematical models, such as decision trees with different number of branches (tree: simple and medium); support vector machines with different kernel functions [SVM: linear and radial basis function (RBF)]; <italic>k</italic>-nearest neighbor classification with Euclidean distance and three different numbers of nearest neighbors, a weighted distance function and a cosine distance metric (KNN: fine, medium, coarse, weighted, and cosine); and discriminant analyses with linear or quadratic decision surface including linear diagonal or quadratic diagonal models (discriminant: linear, quadratic, diaglinear, and diagquadratic). These models are available in the MATLAB Classification Toolbox.</p>
<p>As a starting point, we tested the performance of using the CD amplitudes at 200 and 222&#xa0;nm wavelengths to identify IDPs using the 85 spectra collected from the literature based on Uversky&#x2019;s works (<xref ref-type="bibr" rid="B40">Uversky, 2002</xref>; <xref ref-type="bibr" rid="B42">Uversky, 2003</xref>; <xref ref-type="bibr" rid="B39">Uversky and Fink, 2004</xref>) as training set and using our entire database as test set (in a cross-validated manner). SVM&#x2013;RBF was proven to be the best mathematical model providing 11.1, 3.5, and 8.6% errors in identifying the ordered structures, disordered structures, in overall accuracy, respectively (see also <xref ref-type="fig" rid="F1">Figure 1B</xref>). In the next step, we tested the performance of all models using CD data at two wavelengths varying the wavelength values to find the best performing pairs as a function of the cutoff wavelength of the CD spectra. The different methods varied in global error and in the error on ordered and disordered structures. We selected the best methods for minimal global errors and for minimal errors in disorder prediction. The results were dependent on the spectral range (wavelength cutoff), as shown in <xref ref-type="sec" rid="s10">Supplementary Table S2</xref>. Generally, decision tree algorithms provided good performance; however, other models also gave similar results. The error was increasing with higher cutoff wavelengths. As an example, with 200&#xa0;nm cutoff, SVM-linear showed 7.7 and 2.5% errors for ordered and disordered structures and 6.1% global error using the 204 and 215&#xa0;nm wavelength pair, respectively.</p>
<p>On further analysis, we studied if disorder&#x2013;order classification can be improved by using three data points. Spectra with 175&#xa0;nm cutoff could be classified without any error by the SVM&#x2013;RBF algorithm using the &#x201c;182-194&#x2013;209&#xa0;nm&#x201d; data triplet (<xref ref-type="table" rid="T1">Table 1</xref>). It is worthy to note that the number of disordered spectra was only 21 in this wavelength range. For all algorithms, the error was increasing with higher wavelength cutoff; however, it was significantly lower in the case using two wavelengths for classification. At each cutoff wavelengths, 3&#x2013;5 algorithms gave similar results, making it difficult to select between them at first sight. Generally, SVM-linear and RBF, KNN-fine and cosine, tree-medium, and discriminant-quadratic algorithms using various wavelength triplets worked efficiently. At 200&#xa0;nm cutoff, the accuracy is decreased, which, we believe, is because the spectra collected down to 175 or 180&#xa0;nm have higher quality than the spectra collected from the literature with 190 or 200&#xa0;nm wavelength cutoffs. Spectra in the PCDDB and collected by us underwent a careful inspection (<xref ref-type="bibr" rid="B47">Woollett et al., 2013</xref>). However, the error of the classification is still sufficiently low for these methods to be suitable as experimental classifiers for IDPs (<xref ref-type="table" rid="T1">Table 1</xref>). The error of classification for all the algorithms as the function of cutoff wavelength for two and three wavelengths is presented in <xref ref-type="sec" rid="s10">Supplementary Figures S2&#x2013;S5</xref>. Tables presenting the detailed results of all algorithms are provided as the <xref ref-type="sec" rid="s10">Supplementary Material</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Disorder&#x2013;order classification using three wavelengths.<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th colspan="4" align="center">Wavelength (nm)</th>
<th colspan="3" align="center">Error (%)</th>
</tr>
<tr>
<th align="center">Cutoff (nm)</th>
<th align="center">Algorithm</th>
<th align="center">WL1</th>
<th align="center">WL2</th>
<th align="center">WL3</th>
<th align="center">Ordered</th>
<th align="center">Disordered</th>
<th align="center">Global</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="left">175</td>
<td align="left">SVM&#x2013;RBF</td>
<td align="char" char=".">182</td>
<td align="char" char=".">194</td>
<td align="char" char=".">209</td>
<td align="char" char=".">0</td>
<td align="char" char=".">0</td>
<td align="char" char=".">0</td>
</tr>
<tr>
<td align="left">Discr-quadratic</td>
<td align="char" char=".">179</td>
<td align="char" char=".">214</td>
<td align="char" char=".">225</td>
<td align="char" char=".">0.8</td>
<td align="char" char=".">0</td>
<td align="char" char=".">0.7</td>
</tr>
<tr>
<td align="left">Tree-medium</td>
<td align="char" char=".">192</td>
<td align="char" char=".">220</td>
<td align="char" char=".">228</td>
<td align="char" char=".">0.8</td>
<td align="char" char=".">0</td>
<td align="char" char=".">0.7</td>
</tr>
<tr>
<td rowspan="4" align="left">180</td>
<td align="left">KNN-fine</td>
<td align="char" char=".">184</td>
<td align="char" char=".">197</td>
<td align="char" char=".">208</td>
<td align="char" char=".">0.7</td>
<td align="char" char=".">0</td>
<td align="char" char=".">0.6</td>
</tr>
<tr>
<td align="left">Discr-quadratic</td>
<td align="char" char=".">197</td>
<td align="char" char=".">216</td>
<td align="char" char=".">221</td>
<td align="char" char=".">1.3</td>
<td align="char" char=".">0</td>
<td align="char" char=".">1.1</td>
</tr>
<tr>
<td align="left">SVM&#x2013;RBF</td>
<td align="char" char=".">195</td>
<td align="char" char=".">217</td>
<td align="char" char=".">227</td>
<td align="char" char=".">2</td>
<td align="char" char=".">0</td>
<td align="char" char=".">1.7</td>
</tr>
<tr>
<td align="left">Tree-simple</td>
<td align="char" char=".">185</td>
<td align="char" char=".">192</td>
<td align="char" char=".">211</td>
<td align="char" char=".">2</td>
<td align="char" char=".">0</td>
<td align="char" char=".">1.7</td>
</tr>
<tr>
<td rowspan="3" align="left">185</td>
<td align="left">Tree-medium</td>
<td align="char" char=".">191</td>
<td align="char" char=".">201</td>
<td align="char" char=".">250</td>
<td align="char" char=".">1.3</td>
<td align="char" char=".">2.7</td>
<td align="char" char=".">1.6</td>
</tr>
<tr>
<td align="left">SVM&#x2013;RBF</td>
<td align="char" char=".">195</td>
<td align="char" char=".">217</td>
<td align="char" char=".">227</td>
<td align="char" char=".">2</td>
<td align="char" char=".">2.4</td>
<td align="char" char=".">2.1</td>
</tr>
<tr>
<td align="left">Discr-quadratic</td>
<td align="char" char=".">199</td>
<td align="char" char=".">213</td>
<td align="char" char=".">234</td>
<td align="char" char=".">2</td>
<td align="char" char=".">2.4</td>
<td align="char" char=".">2.1</td>
</tr>
<tr>
<td rowspan="3" align="left">190</td>
<td align="left">Tree-medium</td>
<td align="char" char=".">191</td>
<td align="char" char=".">201</td>
<td align="char" char=".">250</td>
<td align="char" char=".">1.9</td>
<td align="char" char=".">5.6</td>
<td align="char" char=".">2.8</td>
</tr>
<tr>
<td align="left">SVM&#x2013;RBF</td>
<td align="char" char=".">196</td>
<td align="char" char=".">216</td>
<td align="char" char=".">229</td>
<td align="char" char=".">2.4</td>
<td align="char" char=".">5.1</td>
<td align="char" char=".">3.1</td>
</tr>
<tr>
<td align="left">Discr-quadratic</td>
<td align="char" char=".">199</td>
<td align="char" char=".">213</td>
<td align="char" char=".">234</td>
<td align="char" char=".">3.5</td>
<td align="char" char=".">1.7</td>
<td align="char" char=".">3.1</td>
</tr>
<tr>
<td rowspan="5" align="left">195</td>
<td align="left">Discr-quadratic</td>
<td align="char" char=".">199</td>
<td align="char" char=".">213</td>
<td align="char" char=".">234</td>
<td align="char" char=".">3.5</td>
<td align="char" char=".">2.9</td>
<td align="char" char=".">3.3</td>
</tr>
<tr>
<td align="left">SVM-linear</td>
<td align="char" char=".">196</td>
<td align="char" char=".">212</td>
<td align="char" char=".">235</td>
<td align="char" char=".">4.1</td>
<td align="char" char=".">1.5</td>
<td align="char" char=".">3.4</td>
</tr>
<tr>
<td align="left">KNN-cosine</td>
<td align="char" char=".">197</td>
<td align="char" char=".">206</td>
<td align="char" char=".">233</td>
<td align="char" char=".">4.7</td>
<td align="char" char=".">1.5</td>
<td align="char" char=".">3.8</td>
</tr>
<tr>
<td align="left">SVM&#x2013;RBF</td>
<td align="char" char=".">196</td>
<td align="char" char=".">216</td>
<td align="char" char=".">223</td>
<td align="char" char=".">3.5</td>
<td align="char" char=".">4.4</td>
<td align="char" char=".">3.8</td>
</tr>
<tr>
<td align="left">Discr-linear</td>
<td align="char" char=".">195</td>
<td align="char" char=".">219</td>
<td align="char" char=".">237</td>
<td align="char" char=".">3.5</td>
<td align="char" char=".">4.5</td>
<td align="char" char=".">3.8</td>
</tr>
<tr>
<td rowspan="5" align="left">200</td>
<td align="left">KNN-cosine</td>
<td align="char" char=".">212</td>
<td align="char" char=".">217</td>
<td align="char" char=".">225</td>
<td align="char" char=".">4.7</td>
<td align="char" char=".">1.5</td>
<td align="char" char=".">3.8</td>
</tr>
<tr>
<td align="left">SVM-linear</td>
<td align="char" char=".">202</td>
<td align="char" char=".">205</td>
<td align="char" char=".">231</td>
<td align="char" char=".">7.2</td>
<td align="char" char=".">2.5</td>
<td align="char" char=".">5.7</td>
</tr>
<tr>
<td align="left">SVM&#x2013;RBF</td>
<td align="char" char=".">206</td>
<td align="char" char=".">212</td>
<td align="char" char=".">229</td>
<td align="char" char=".">5</td>
<td align="char" char=".">7.5</td>
<td align="char" char=".">5.7</td>
</tr>
<tr>
<td align="left">Discr-quadratic</td>
<td align="char" char=".">201</td>
<td align="char" char=".">211</td>
<td align="char" char=".">215</td>
<td align="char" char=".">6.6</td>
<td align="char" char=".">3.8</td>
<td align="char" char=".">5.7</td>
</tr>
<tr>
<td align="left">KNN-fine</td>
<td align="char" char=".">212</td>
<td align="char" char=".">215</td>
<td align="char" char=".">227</td>
<td align="char" char=".">3.9</td>
<td align="char" char=".">10</td>
<td align="char" char=".">5.7</td>
</tr>
<tr>
<td rowspan="3" align="left">205</td>
<td align="left">KNN-cosine</td>
<td align="char" char=".">212</td>
<td align="char" char=".">217</td>
<td align="char" char=".">225</td>
<td align="char" char=".">3.3</td>
<td align="char" char=".">7.4</td>
<td align="char" char=".">4.6</td>
</tr>
<tr>
<td align="left">SVM&#x2013;RBF</td>
<td align="char" char=".">206</td>
<td align="char" char=".">212</td>
<td align="char" char=".">229</td>
<td align="char" char=".">5</td>
<td align="char" char=".">7.4</td>
<td align="char" char=".">5.7</td>
</tr>
<tr>
<td align="left">KNN-fine</td>
<td align="char" char=".">212</td>
<td align="char" char=".">215</td>
<td align="char" char=".">227</td>
<td align="char" char=".">3.9</td>
<td align="char" char=".">9.9</td>
<td align="char" char=".">5.7</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>a</label>
<p>Algorithms showing the least errors using three wavelengths (WL1, WL2, WL3) for classification as a function of the cutoff wavelength are presented. For training dataset, for a given wavelength triplet, all proteins&#x2019; spectra that covered those wavelengths were used.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-4">
<title>Effect of Concentration Error on Disorder&#x2013;Order Classification</title>
<p>Due to their unusual amino acid composition, concentration determination of IDPs with the widely used basic techniques is challenging and might lead to large inaccuracies (<xref ref-type="bibr" rid="B36">Sz&#x151;ll&#x151;si et al., 2007</xref>). Measurement by the aromatic absorption is problematic because of the usually low number of such residues in IDPs. Colorimetric assays are also affected by the special amino acid composition of IDPs and are sensitive to contaminations. One solution might be the absorbance measurement at 205 or 214&#xa0;nm (<xref ref-type="bibr" rid="B17">Kuipers and Gruppen, 2007</xref>; <xref ref-type="bibr" rid="B2">Anthis and Clore, 2013</xref>; <xref ref-type="bibr" rid="B20">Micsonai et al., 2021</xref>); however, buffer absorption can limit its applicability. Measurement by mass of the dry sample usually also produces errors because of the bound water or remaining salts. We estimated that a 20% error might regularly occur in concentration measurements of IDPs, which might have an effect on the accuracy of disorder classification. Thus, we tested the robustness of the classification algorithms for such errors by re-evaluating the spectra after rescaling them with factors between 0.5 and 2. <xref ref-type="sec" rid="s10">Supplementary Figure S6</xref> shows the dependence of the classification error on the rescaling for the various algorithms presented in <xref ref-type="table" rid="T1">Table 1</xref>. Most of the methods showed a surprisingly high sensitivity for concentration errors. The SVM&#x2013;RBF algorithm works without error on the correctly normalized spectra (scale factor &#x3d; 1); however, even a 10% increase in the spectral amplitude increases the error on the disordered structure identification to over 10% (<xref ref-type="fig" rid="F3">Figure 3A</xref>). The exception is the KNN-cosine method, which showed no dependence on the spectral amplitude (<xref ref-type="fig" rid="F3">Figure 3A</xref>). The &#x201c;cosine&#x201d; distance metric of the KNN algorithm uses the cosine of the angle between vectors pointing from the origin and data points. The direction of these vectors will neither change with scaling nor will the angles. Considering these facts, we propose the selection of KNN-cosine as the optimal classification algorithm. It performs with acceptable accuracy and is free of concentration errors (<xref ref-type="fig" rid="F3">Figures 3B,C</xref>). <xref ref-type="table" rid="T2">Table 2</xref> shows the performance of KNN-cosine as a function of the wavelength cutoff. Intriguingly, the best wavelength triplet in the cutoff range from 175 to 179&#xa0;nm is proven to be the &#x201c;214-218&#x2013;232&#xa0;nm&#x201d; triplet. It suggests that we do not really need CD data down to 175&#xa0;nm for the binary classification. However, with a cutoff of 200&#xa0;nm, KNN-cosine provided significantly lower accuracy, despite the fact that the lower wavelength range was not needed for the method. We believe this is because of the quality difference between SRCD spectra collected down to 175&#xa0;nm and conventional measurements with 200&#xa0;nm wavelength minimum. To address this question, further investigations were carried out.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Effect of concentration error on disordered&#x2013;ordered classification and introduction of the KNN-cosine method. <bold>(A)</bold> Error of the SVM&#x2013;RBF algorithm as a function of the scaling factor on the spectra of the database with 175&#xa0;nm cutoff are shown for disordered (red) and ordered (black) structures. The global error is shown in blue. Dashed lines show the errors of classification using the KNN-cosine algorithm for disordered (red), ordered (black), and the overall error (blue). For convenience, &#xb1;20% and &#xb1;50% changes in the concentration (i.e., in the scaling factor) are shown. <bold>(B)</bold> Reference points in the space determined by the CD data measured at 197, 206, and 233&#xa0;nm wavelengths and an example for vectors by using the KNN-cosine method. Red and blue points represent ordered and disordered proteins, respectively. <bold>(C)</bold> The distance metric of this KNN algorithm uses the cosine of the angle between vectors pointing from the origin to data points. The prediction is based on the labels (ordered/disordered) of the first 10 reference points with the lowest &#x201c;distance&#x201d; from the test point. The direction and the angles of the vectors will not change with scaling, that is, the method is independent of concentration errors.</p>
</caption>
<graphic xlink:href="fmolb-09-863141-g003.tif"/>
</fig>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Accuracy of KNN-cosine algorithm as a function of cutoff wavelength.<xref ref-type="table-fn" rid="Tfn2">
<sup>a</sup>
</xref>
</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th colspan="3" align="center">Wavelength (nm)</th>
<th colspan="3" align="center">Error (%)</th>
</tr>
<tr>
<th align="center">Cutoff (nm)</th>
<th align="center">WL1</th>
<th align="center">WL2</th>
<th align="center">WL3</th>
<th align="center">Ordered</th>
<th align="center">Disordered</th>
<th align="center">Global</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">175</td>
<td align="char" char=".">214</td>
<td align="char" char=".">218</td>
<td align="char" char=".">232</td>
<td align="char" char=".">1.6</td>
<td align="char" char=".">0</td>
<td align="char" char=".">1.3</td>
</tr>
<tr>
<td align="left">176</td>
<td align="char" char=".">214</td>
<td align="char" char=".">218</td>
<td align="char" char=".">232</td>
<td align="char" char=".">1.5</td>
<td align="char" char=".">0</td>
<td align="char" char=".">1.3</td>
</tr>
<tr>
<td align="left">177</td>
<td align="char" char=".">214</td>
<td align="char" char=".">218</td>
<td align="char" char=".">232</td>
<td align="char" char=".">1.5</td>
<td align="char" char=".">0</td>
<td align="char" char=".">1.3</td>
</tr>
<tr>
<td align="left">178</td>
<td align="char" char=".">214</td>
<td align="char" char=".">218</td>
<td align="char" char=".">232</td>
<td align="char" char=".">1.5</td>
<td align="char" char=".">0</td>
<td align="char" char=".">1.3</td>
</tr>
<tr>
<td align="left">179</td>
<td align="char" char=".">214</td>
<td align="char" char=".">218</td>
<td align="char" char=".">232</td>
<td align="char" char=".">1.5</td>
<td align="char" char=".">0</td>
<td align="char" char=".">1.3</td>
</tr>
<tr>
<td align="left">180</td>
<td align="char" char=".">197</td>
<td align="char" char=".">206</td>
<td align="char" char=".">233</td>
<td align="char" char=".">4</td>
<td align="char" char=".">0</td>
<td align="char" char=".">3.4</td>
</tr>
<tr>
<td align="left">183</td>
<td align="char" char=".">197</td>
<td align="char" char=".">206</td>
<td align="char" char=".">233</td>
<td align="char" char=".">4</td>
<td align="char" char=".">0</td>
<td align="char" char=".">3.3</td>
</tr>
<tr>
<td align="left">185</td>
<td align="char" char=".">197</td>
<td align="char" char=".">206</td>
<td align="char" char=".">233</td>
<td align="char" char=".">3.9</td>
<td align="char" char=".">0</td>
<td align="char" char=".">3.1</td>
</tr>
<tr>
<td align="left">190</td>
<td align="char" char=".">197</td>
<td align="char" char=".">206</td>
<td align="char" char=".">233</td>
<td align="char" char=".">4.7</td>
<td align="char" char=".">1.7</td>
<td align="char" char=".">3.9</td>
</tr>
<tr>
<td align="left">195</td>
<td align="char" char=".">197</td>
<td align="char" char=".">206</td>
<td align="char" char=".">233</td>
<td align="char" char=".">4.7</td>
<td align="char" char=".">1.5</td>
<td align="char" char=".">3.8</td>
</tr>
<tr>
<td align="left">198</td>
<td align="char" char=".">198</td>
<td align="char" char=".">205</td>
<td align="char" char=".">237</td>
<td align="char" char=".">4</td>
<td align="char" char=".">2.9</td>
<td align="char" char=".">3.7</td>
</tr>
<tr>
<td align="left">200</td>
<td align="char" char=".">212</td>
<td align="char" char=".">217</td>
<td align="char" char=".">225</td>
<td align="char" char=".">3.3</td>
<td align="char" char=".">7.5</td>
<td align="char" char=".">4.6</td>
</tr>
<tr>
<td align="left">205</td>
<td align="char" char=".">212</td>
<td align="char" char=".">217</td>
<td align="char" char=".">225</td>
<td align="char" char=".">3.3</td>
<td align="char" char=".">7.4</td>
<td align="char" char=".">4.6</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn2">
<label>a</label>
<p>Wavelengths of the data points (WL1, WL2, WL3) for the best performance at each cutoff and the errors of classification are shown.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-5">
<title>Effect of Experimental Noise</title>
<p>To investigate the effect of spectrum quality/spectral noise on the disordered&#x2013;ordered classification, we added artificial noise to the spectra tested. The noise was added independently to each CD value using random values from normal distribution (&#xb5; &#x3d; 0&#xa0;M<sup>&#x2212;1</sup> cm<sup>&#x2212;1</sup>, &#x3c3; &#x3d; 0.1&#xa0;M<sup>&#x2212;1</sup> cm<sup>&#x2212;1</sup> or &#x3c3; &#x3d; 0.05&#xa0;M<sup>&#x2212;1</sup> cm<sup>&#x2212;1</sup>). The effect of noise was calculated by averaging the results of 1,000 simulations on each of the wavelength triplets of the KNN-cosine model on the possible wavelength cutoff ranges (<xref ref-type="sec" rid="s10">Supplementary Figure S7</xref>). The addition of noise significantly increased the error of classification. Noise had the highest effect when using the &#x201c;214-218-232&#xa0;nm&#x201d; data triplet possibly because 214 and 218&#xa0;nm data are close to each other. The &#x201c;197-206-233&#xa0;nm&#x201d; triplet was more robust for noise and generally showed a good performance for all possible wavelength ranges from 175&#xa0;nm up to 197&#xa0;nm cutoffs. Therefore, we suggest using this model as a classification tool. Above 197&#xa0;nm, the &#x201c;212-217-225&#xa0;nm&#x201d; data triplet should be used (<xref ref-type="sec" rid="s10">Supplementary Figure S7</xref>). Performance of the KNN-cosine method combined for all wavelength cutoffs including the effect of noise is presented in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Accuracy of the KNN-cosine method as a function of wavelength cutoff. Error on disordered (red) and ordered (black) proteins and the global error (blue) are shown with solid curves for the original spectra and with dashed and dotted lines for spectra with added noise of &#x3c3; &#x3d; 0.05 and 0.1&#xa0;M<sup>&#x2212;1</sup>cm<sup>&#x2212;1</sup>, respectively. Up to 197&#xa0;nm cutoff, the &#x201c;197-206-233&#xa0;nm&#x201d; triplet and above 197&#xa0;nm, the &#x201c;212-217-225&#xa0;nm&#x201d; triplet were used for analysis.</p>
</caption>
<graphic xlink:href="fmolb-09-863141-g004.tif"/>
</fig>
<p>Based on all these results, for the best disorder&#x2013;order classification, it is recommended to collect good-quality CD spectra down to &#x223c;195&#xa0;nm and use the KNN-cosine algorithm with data at 197-206-233&#xa0;nm wavelengths.</p>
</sec>
<sec id="s3-6">
<title>Disorder Classification for Limited Wavelength Range, Under Strong Absorbing Conditions</title>
<p>It is an interesting and maybe unexpected finding that KNN-cosine with the data triplet &#x201c;212-217-225&#xa0;nm,&#x201d; that is, with 212&#xa0;nm lowest wavelength, is a good choice for disorder classification. Although the error shown in <xref ref-type="fig" rid="F4">Figure 4</xref> is increased for cutoffs above 197&#xa0;nm, this is somewhat misleading. This method gives better results for high-quality spectra recorded down to 175&#x2013;180&#xa0;nm even without using any of their data points below 212&#xa0;nm for the classification (<xref ref-type="sec" rid="s10">Supplementary Figure S7</xref>). The error on these spectra, downloaded from PCDDB or measured by us using SRCD, is 3.3, 0, and 2.8% for ordered structure, disordered structure, and globally, respectively. These spectra were treated and validated using careful protocols (<xref ref-type="bibr" rid="B14">Kelly et al., 2005</xref>; <xref ref-type="bibr" rid="B47">Woollett et al., 2013</xref>; <xref ref-type="bibr" rid="B20">Micsonai et al., 2021</xref>). The 89 spectra collected from the literature have obviously lower average quality, and this increases the error of the classification on them to 3.3, 10.9, and 8.24% for disordered and ordered structures and for global error, respectively. These calculations were performed in a leave-one-out cross-validated manner using all available data as training dataset. Careful, noiseless experiments with correct baseline subtractions might give better accuracy than the average error found here.</p>
<p>A real advantage of the KNN-cosine method with &#x201c;212-217-225&#xa0;nm&#x201d; data is that it can be used for CD spectra recorded in the presence of strongly absorbing solutions such as the case of chemical denaturants (e.g., urea and GdnHCl), or under crowded conditions if the spectrum can only be recorded down to &#x223c;210&#xa0;nm. It might help to study the crucial question if a supposedly IDP will indeed exhibit disordered structures under crowded conditions or become structured (<xref ref-type="bibr" rid="B35">Szasz et al., 2011</xref>; <xref ref-type="bibr" rid="B30">Qin and Zhou, 2013</xref>; <xref ref-type="bibr" rid="B3">Banks et al., 2018</xref>; <xref ref-type="bibr" rid="B33">Simpson et al., 2020</xref>; <xref ref-type="bibr" rid="B16">K&#xf6;nig et al., 2021</xref>).</p>
</sec>
<sec id="s3-7">
<title>Experimental Classification of Disorder vs. <italic>In Silico</italic> Predictions</title>
<p>Numerous bioinformatics tools have been developed in the last decade to predict intrinsic disorder from the amino acid sequence (<xref ref-type="bibr" rid="B12">Katuwawala et al., 2019</xref>; <xref ref-type="bibr" rid="B18">Liu et al., 2019</xref>; <xref ref-type="bibr" rid="B25">Necci et al., 2021</xref>). Among them, AlphaFold2 was proven to be the most accurate method to detect disorder. Low values of the plDDT parameter (confidence) have been shown to be indicative of disordered regions (<xref ref-type="bibr" rid="B10">Jumper et al., 2021</xref>)<italic>.</italic> AlphaFold2 and previous methods are useful when investigating large datasets, and high-throughput analysis is needed, and they indeed provide good statistics. However, <italic>in silico</italic> predictions always have a level of uncertainty and thus need experimental verification, especially when investigations are narrowed down and focus on a particular protein. To confirm this statement, we analyzed the disordered proteins of our reference database by AlphaFold2 and found that several disordered chains were mistakenly predicted to be highly &#x3b1;-helical, such as &#x3b1;-synuclein, thymosin-&#x3b1;1, basic subdomain of the c-Jun oncoprotein, &#x3b1;-tubulin (fragment 404&#x2013;451), &#x3b2;-tubulin (fragment 395&#x2013;445), S21 protein from the 30S subunit of the <italic>E. coli</italic> ribosome, and artificial disordered peptides &#x23;1, 2, and 6. Moreover, computational methods, like AlphaFold2, can neither take the actual environmental conditions into account, such as pH, ionic strength, temperature, the presence of additives or crowding agents, the effect of protein concentration, intermolecular interactions, nor accurately calculate the effect of single mutations (unless the crystal structure was already solved and deposited in the PDB) and the effects of post-translational modifications (e.g., phosphorylation) (<xref ref-type="bibr" rid="B26">Pak et al., 2021</xref>; <xref ref-type="bibr" rid="B28">Perrakis and Sixma, 2021</xref>). As IDPs are specifically sensitive to their surroundings, depending on the solvent environment, a single polypeptide chain can take up various conformations, which results in important biological readouts. Therefore, an experimental method, such as CD spectroscopy, can validate and specify the prediction of AlphaFold2 and should be used for this purpose. When CD spectroscopy confirms the prediction of AlphaFold2, the site-specific information of AlphaFold2 is likely valuable. However, if there is a large discrepancy between the prediction of AlphaFold2 and the experimental results, then the priority has to be given to the experience.</p>
</sec>
<sec id="s3-8">
<title>Case Studies</title>
<p>As a further support for the aforementioned statement, here, we provide specific case studies presenting the dependence of the protein structure and disorder on the buffer conditions. These reveal the necessity of experimental techniques and the limitations of <italic>in silico</italic> predictions for the detection of protein disorder. One example is the well-known &#x3b1;-synuclein, a protein associated with Parkinson&#x2019;s disease. It is an IDP, and CD spectroscopy shows that indeed, the protein is disordered under physiological buffer conditions. In the presence of 30% TFE, which mimics a less polar solvent environment, such as in membranes, the protein becomes ordered with 47% &#x3b1;-helix content as estimated from the CD spectrum by the BeStSel algorithm (<xref ref-type="bibr" rid="B22">Micsonai et al., 2015</xref>; <xref ref-type="bibr" rid="B21">Micsonai et al., 2018</xref>). At concentrations above 2&#xa0;mg/ml, &#x3b1;-synuclein readily forms oligomers in 30% TFE with a spectral shape characteristic of the &#x3b2;-structure. The corresponding CD spectra of &#x3b1;-synuclein and results of the binary classification are shown in <xref ref-type="fig" rid="F5">Figure 5A</xref>. In contrast, AlphaFold2, irrespectively of the buffer conditions, erroneously predicts with high confidence that 64% of the &#x3b1;-synuclein chain is in an &#x3b1;-helical structure.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Case studies showing the structural variability of individual proteins, which can only be revealed experimentally. <bold>(A)</bold> CD spectrum of &#x3b1;-synuclein in water is characteristic of a fully disordered chain. In 30% TFE, the protein exhibits an ordered, &#x3b1;-helix-rich conformation, and at higher concentrations (10&#xa0;mg/ml), it readily forms oligomers with a spectral shape of &#x3b2;-structure. <bold>(B)</bold> In the native state, &#x3b2;<sub>2</sub>-microglobulin (&#x3b2;2m) exhibits a &#x3b2;-sandwich fold of an antiparallel &#x3b2;-structure. At low pH or in 3&#xa0;M GdnHCl, its structure becomes disordered. <bold>(C)</bold> ERD14 disordered plant chaperone and its artificial pair consisting of the full scrambled sequence both exhibit disordered structure in water. The presence of 30% TFE induces the formation of &#x3b1;-helix in the wild-type protein, while its scrambled variant preserves its disordered conformation. For <bold>(C)</bold>, experimental data modified from <xref ref-type="bibr" rid="B23">Murvai et al. (2021)</xref> were used with the authors&#x2019; permission. The results of the binary classification are shown by O (ordered) and D (disordered) letters in the figures.</p>
</caption>
<graphic xlink:href="fmolb-09-863141-g005.tif"/>
</fig>
<p>&#x3b2;2-microglobulin (&#x3b2;2m) is the light chain of MHC-1 and can also be found in a monomeric form in the blood. It causes serious complications upon long-term dialysis depositing in the form of amyloid fibrils in the osteoarticular system of patients. The native protein exhibits an immunoglobulin fold with an antiparallel &#x3b2;-sandwich, which is a cinch for AlphaFold2. However, &#x3b2;2m is sensitive to the drop of pH; it becomes unfolded below pH 4, which cannot be deduced from the AlphaFold2 prediction. We also present the disordered spectrum of the protein in 3&#xa0;M GdnHCl, showing that it is possible to identify disordered structures even in highly absorbing solutions by our method (<xref ref-type="fig" rid="F5">Figure 5B</xref>).</p>
<p>ERD14 is a disordered plant chaperone, which is correctly predicted by AlphaFold2. However, in 30% TFE, the protein gains a significant amount of &#x3b1;-helix, which turns out to be indispensable for the protein&#x2019;s function (<xref ref-type="bibr" rid="B24">Murvai et al., 2020</xref>). An artificial variant of ERD14 with a full-scrambled sequence (having the same amino acid composition) and no biological function is similarly disordered in water; however, in the presence of TFE, it still preserves its disordered conformation. In this type of comparison, CD spectroscopy reveals the secondary structure forming tendency of disordered wild-type ERD14 under suitable conditions or upon intermolecular interactions (<xref ref-type="fig" rid="F5">Figure 5C</xref>).</p>
<p>In our previous work on a Trp-cage miniprotein (<xref ref-type="bibr" rid="B11">Kardos et al., 2015</xref>), we showed that a single side-chain phosphorylation can cause drastic conformational changes. Our classification shows that the protein obviously loses its &#x3b1;-helix content and becomes disordered upon the phosphorylation of its Ser9 residue. Such drastic change is also missed when the structure is predicted with AlphaFold2.</p>
<p>These examples reveal the limitations of <italic>in silico</italic> predictions and the necessity of integration of various experimental techniques for the detection of protein disorder.</p>
</sec>
<sec id="s3-9">
<title>Limitations: Intrinsically Disordered Regions (IDRs)</title>
<p>The binary classification method presented here is to identify essentially disordered proteins, that is, to detect &#x201c;global&#x201d; disorder. In the case of partial disorder, this binary classification will not detect a disordered protein region of an otherwise ordered protein. In such a case, partial disorder can be deduced from the secondary structure composition determined by analyzing the entire CD spectrum with some of the available methods, such as BeStSel (<xref ref-type="bibr" rid="B21">Micsonai et al., 2018</xref>; <xref ref-type="bibr" rid="B20">Micsonai et al., 2021</xref>). Upon intermolecular interactions of disordered proteins, localized segments might take up ordered structure, which, depending on the size of the segment, might not change the result of the classification. To study such partial structural changes, a full CD spectrum analysis is required with BeStSel (<xref ref-type="bibr" rid="B22">Micsonai et al., 2015</xref>; <xref ref-type="bibr" rid="B21">Micsonai et al., 2018</xref>) or other algorithms (<xref ref-type="bibr" rid="B34">Sreerama and Woody, 2000</xref>; <xref ref-type="bibr" rid="B19">Lobley et al., 2002</xref>).</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>Conclusion</title>
<p>Intrinsically disordered proteins are abundant in nature and responsible for a plethora of cellular functions (<xref ref-type="bibr" rid="B6">Dunker et al., 2002</xref>; <xref ref-type="bibr" rid="B7">Habchi et al., 2014</xref>). They lack a stable tertiary structure and form dynamic conformational ensembles due to their characteristic physicochemical properties and amino acid composition (<xref ref-type="bibr" rid="B43">Varadi et al., 2015</xref>; <xref ref-type="bibr" rid="B13">Katuwawala et al., 2020</xref>). Although numerous bioinformatics tools have been developed for disorder prediction in the last 2&#xa0;decades, there is still a high need for experimental verification of the disordered state. Here, we proposed an automatized binary disorder&#x2013;order classification by analyzing far-UV CD spectroscopy data. The method uses CD data at three wavelength points, which makes high-throughput data collection possible. To reach the best classification accuracy, CD of the protein should be measurable down to 197&#xa0;nm in good quality. However, in case of strong absorbing samples, such as in crowded environmental conditions, 212&#xa0;nm lowest wavelength still provides acceptable performance. The mathematical analysis uses the <italic>k</italic>-nearest neighbor algorithm with cosine distance function, which is independent of the spectral amplitude, that is, free of concentration determination errors. We believe the classification method will be useful in identifying or verifying disorder in individual problems and will also facilitate the growth of experimental data in IDP databases, such as DisProt (<xref ref-type="bibr" rid="B31">Quaglia et al., 2021</xref>). The method is implemented on a webserver and freely available for academic use at <ext-link ext-link-type="uri" xlink:href="https://bestsel.elte.hu/idp_classification.php">https://bestsel.elte.hu/idp_classification.php</ext-link>.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The datasets generated for this study are included in the article/<xref ref-type="sec" rid="s10">Supplementary Material</xref>; further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>AM, &#xc9;M, NM, FW, OT, &#xc1;T, and JK performed experiments. AM, &#xc9;M, and JK analyzed results. MR and JK supervised experiments. AM and JK designed the work. AM, &#xc9;M, and JK wrote the manuscript. All authors read and approved the final manuscript.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This study was supported by the National Research, Development and Innovation Office of Hungary (grants 2017-1.2.1-NKP-2017-00002, PD135510, K120391, K125340, K131702, K138937 and 2019-2.1.11-T&#x00C9;T-2020-00101). SRCD measurements were supported by SOLEIL (Proposals 20181890, 20191810, and 20200751).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>We thank Zsuzsanna Doszt&#xe1;nyi for the enlightening discussions.</p>
</ack>
<sec id="s10">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmolb.2022.863141/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmolb.2022.863141/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet2.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet3.xlsx" id="SM2" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM3" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adler</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Greenfield</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Fasman</surname>
<given-names>G. D.</given-names>
</name>
</person-group> (<year>1973</year>). <article-title>[27] Circular Dichroism and Optical Rotatory Dispersion of Proteins and Polypeptides</article-title>. <source>Methods Enzymol.</source> <volume>27</volume>, <fpage>675</fpage>&#x2013;<lpage>735</lpage>. <pub-id pub-id-type="doi">10.1016/s0076-6879(73)27030-1</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anthis</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Clore</surname>
<given-names>G. M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Sequence-specific Determination of Protein and Peptide Concentrations by Absorbance at 205 Nm</article-title>. <source>Protein Sci.</source> <volume>22</volume> (<issue>6</issue>), <fpage>851</fpage>&#x2013;<lpage>858</lpage>. <pub-id pub-id-type="doi">10.1002/pro.2253</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Banks</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Weiss</surname>
<given-names>K. L.</given-names>
</name>
<name>
<surname>Stanley</surname>
<given-names>C. B.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.-X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Intrinsically Disordered Protein Exhibits Both Compaction and Expansion under Macromolecular Crowding</article-title>. <source>Biophysical J.</source> <volume>114</volume> (<issue>5</issue>), <fpage>1067</fpage>&#x2013;<lpage>1079</lpage>. <pub-id pub-id-type="doi">10.1016/j.bpj.2018.01.011</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J. T.</given-names>
</name>
</person-group> (<year>1977</year>). <article-title>Two-Point Calibration of Circular Dichrometer with D-10-Camphorsulfonic Acid</article-title>. <source>Anal. Lett.</source> <volume>10</volume> (<issue>14</issue>), <fpage>1195</fpage>&#x2013;<lpage>1207</lpage>. <pub-id pub-id-type="doi">10.1080/00032717708067855</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dunker</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Obradovic</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Romero</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Garner</surname>
<given-names>E. C.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>C. J.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Intrinsic Protein Disorder in Complete Genomes</article-title>. <source>Genome Inform. Ser. Workshop Genome Inform.</source> <volume>11</volume>, <fpage>161</fpage>&#x2013;<lpage>171</lpage>. </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dunker</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Lawson</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Iakoucheva</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Obradovi&#x107;</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Intrinsic Disorder and Protein Function</article-title>. <source>Biochemistry</source> <volume>41</volume> (<issue>21</issue>), <fpage>6573</fpage>&#x2013;<lpage>6582</lpage>. <pub-id pub-id-type="doi">10.1021/bi012159&#x2b;</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Habchi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tompa</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Longhi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Uversky</surname>
<given-names>V. N.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Introducing Protein Intrinsic Disorder</article-title>. <source>Chem. Rev.</source> <volume>114</volume> (<issue>13</issue>), <fpage>6561</fpage>&#x2013;<lpage>6588</lpage>. <pub-id pub-id-type="doi">10.1021/cr400514h</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ho</surname>
<given-names>B. K.</given-names>
</name>
<name>
<surname>Curmi</surname>
<given-names>P. M. G.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Twist and Shear in &#x3b2;-sheets and &#x3b2;-ribbons</article-title>. <source>J. Mol. Biol.</source> <volume>317</volume> (<issue>2</issue>), <fpage>291</fpage>&#x2013;<lpage>308</lpage>. <pub-id pub-id-type="doi">10.1006/jmbi.2001.5385</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johnson</surname>
<given-names>W. C.</given-names>
<suffix>Jr.</suffix>
</name>
</person-group> (<year>1988</year>). <article-title>Secondary Structure of Proteins through Circular Dichroism Spectroscopy</article-title>. <source>Annu. Rev. Biophys. Biophys. Chem.</source> <volume>17</volume>, <fpage>145</fpage>&#x2013;<lpage>166</lpage>. <pub-id pub-id-type="doi">10.1146/annurev.bb.17.060188.001045</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jumper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Figurnov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ronneberger</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Highly Accurate Protein Structure Prediction with AlphaFold</article-title>. <source>Nature</source> <volume>596</volume> (<issue>7873</issue>), <fpage>583</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kardos</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kiss</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Micsonai</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rov&#xf3;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Menyh&#xe1;rd</surname>
<given-names>D. K.</given-names>
</name>
<name>
<surname>Kov&#xe1;cs</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Phosphorylation as Conformational Switch from the Native to Amyloid State: Trp-Cage as a Protein Aggregation Model</article-title>. <source>J. Phys. Chem. B</source> <volume>119</volume> (<issue>7</issue>), <fpage>2946</fpage>&#x2013;<lpage>2955</lpage>. <pub-id pub-id-type="doi">10.1021/jp5124234</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Katuwawala</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ghadermarzi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kurgan</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Computational Prediction of Functions of Intrinsically Disordered Regions</article-title>. <source>Prog. Mol. Biol. Transl Sci.</source> <volume>166</volume>, <fpage>341</fpage>&#x2013;<lpage>369</lpage>. <pub-id pub-id-type="doi">10.1016/bs.pmbts.2019.04.006</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Katuwawala</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Oldfield</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Kurgan</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Accuracy of Protein-Level Disorder Predictions</article-title>. <source>Brief Bioinform</source> <volume>21</volume> (<issue>5</issue>), <fpage>1509</fpage>&#x2013;<lpage>1522</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbz100</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kelly</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Jess</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Price</surname>
<given-names>N. C.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>How to Study Proteins by Circular Dichroism</article-title>. <source>Biochim. Biophys. Acta (Bba) - Proteins Proteomics</source> <volume>1751</volume> (<issue>2</issue>), <fpage>119</fpage>&#x2013;<lpage>139</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbapap.2005.06.005</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kelly</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Price</surname>
<given-names>N. C.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>The Application of Circular Dichroism to Studies of Protein Folding and Unfolding</article-title>. <source>Biochim. Biophys. Acta (Bba) - Protein Struct. Mol. Enzymol.</source> <volume>1338</volume> (<issue>2</issue>), <fpage>161</fpage>&#x2013;<lpage>185</lpage>. <pub-id pub-id-type="doi">10.1016/s0167-4838(96)00190-2</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>K&#xf6;nig</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Soranno</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nettels</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Schuler</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Impact of In&#x2010;Cell and In&#x2010;Vitro Crowding on the Conformations and Dynamics of an Intrinsically Disordered Protein</article-title>. <source>Angew. Chem. Int. Ed.</source> <volume>60</volume> (<issue>19</issue>), <fpage>10724</fpage>&#x2013;<lpage>10729</lpage>. <pub-id pub-id-type="doi">10.1002/anie.202016804</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuipers</surname>
<given-names>B. J. H.</given-names>
</name>
<name>
<surname>Gruppen</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Prediction of Molar Extinction Coefficients of Proteins and Peptides Using UV Absorption of the Constituent Amino Acids at 214 Nm to Enable Quantitative Reverse Phase High-Performance Liquid Chromatography&#x2212;Mass Spectrometry Analysis</article-title>. <source>J. Agric. Food Chem.</source> <volume>55</volume> (<issue>14</issue>), <fpage>5445</fpage>&#x2013;<lpage>5451</lpage>. <pub-id pub-id-type="doi">10.1021/jf070337l</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A Comprehensive Review and Comparison of Existing Computational Methods for Intrinsically Disordered Protein and Region Prediction</article-title>. <source>Brief Bioinform</source> <volume>20</volume> (<issue>1</issue>), <fpage>330</fpage>&#x2013;<lpage>346</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbx126</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lobley</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Whitmore</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wallace</surname>
<given-names>B. A.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>DICHROWEB: an Interactive Website for the Analysis of Protein Secondary Structure from Circular Dichroism Spectra</article-title>. <source>Bioinformatics</source> <volume>18</volume> (<issue>1</issue>), <fpage>211</fpage>&#x2013;<lpage>212</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/18.1.211</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Micsonai</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Buly&#xe1;ki</surname>
<given-names>&#xc9;.</given-names>
</name>
<name>
<surname>Kardos</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>BeStSel: From Secondary Structure Analysis to Protein Fold Prediction by Circular Dichroism Spectroscopy</article-title>. <source>Methods Mol. Biol.</source> <volume>2199</volume>, <fpage>175</fpage>&#x2013;<lpage>189</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-0716-0892-0_11</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Micsonai</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wien</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Buly&#xe1;ki</surname>
<given-names>&#xc9;.</given-names>
</name>
<name>
<surname>Kun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Moussong</surname>
<given-names>&#xc9;.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>Y.-H.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>BeStSel: a Web Server for Accurate Protein Secondary Structure Prediction and Fold Recognition from the Circular Dichroism Spectra</article-title>. <source>Nucleic Acids Res.</source> <volume>46</volume> (<issue>W1</issue>), <fpage>W315</fpage>&#x2013;<lpage>W322</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky497</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Micsonai</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wien</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kernya</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Goto</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>R&#xe9;fr&#xe9;giers</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Accurate Secondary Structure Prediction and Fold Recognition for Circular Dichroism Spectroscopy</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> <volume>112</volume> (<issue>24</issue>), <fpage>E3095</fpage>&#x2013;<lpage>E3103</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1500851112</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Murvai</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kalmar</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Szabo</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Schad</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Micsonai</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kardos</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Cellular Chaperone Function of Intrinsically Disordered Dehydrin ERD14</article-title>. <source>Ijms</source> <volume>22</volume> (<issue>12</issue>), <fpage>6190</fpage>. <pub-id pub-id-type="doi">10.3390/ijms22126190</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Murvai</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kalmar</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Szalaine Agoston</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Szabo</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Tantos</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Csikos</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Interplay of Structural Disorder and Short Binding Elements in the Cellular Chaperone Function of Plant Dehydrin ERD14</article-title>. <source>Cells</source> <volume>9</volume> (<issue>8</issue>), <fpage>1856</fpage>. <pub-id pub-id-type="doi">10.3390/cells9081856</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Necci</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Piovesan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Piovesan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tosatto</surname>
<given-names>S. C. E.</given-names>
</name>
<name>
<surname>Tosatto</surname>
<given-names>S. C. E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Critical Assessment of Protein Intrinsic Disorder Prediction</article-title>. <source>Nat. Methods</source> <volume>18</volume> (<issue>5</issue>), <fpage>472</fpage>&#x2013;<lpage>481</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-021-01117-3</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pak</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Markhieva</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Novikova</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Petrov</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Vorobyev</surname>
<given-names>I. S.</given-names>
</name>
<name>
<surname>Maksimova</surname>
<given-names>E. S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Using AlphaFold to Predict the Impact of Single Mutations on Protein Stability and Function</article-title>. <source>bioRxiv</source> <volume>2021</volume>, <fpage>460937</fpage>. <pub-id pub-id-type="doi">10.1101/2021.09.19.460937</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Mizianty</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Exceptionally Abundant Exceptions: Comprehensive Characterization of Intrinsic Disorder in All Domains of Life</article-title>. <source>Cell. Mol. Life Sci.</source> <volume>72</volume> (<issue>1</issue>), <fpage>137</fpage>&#x2013;<lpage>151</lpage>. <pub-id pub-id-type="doi">10.1007/s00018-014-1661-9</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Perrakis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sixma</surname>
<given-names>T. K.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>AI Revolutions in Biology</article-title>. <source>EMBO Rep.</source> <volume>22</volume> (<issue>11</issue>), <fpage>e54046</fpage>. <pub-id pub-id-type="doi">10.15252/embr.202154046</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Provencher</surname>
<given-names>S. W.</given-names>
</name>
<name>
<surname>Gloeckner</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1981</year>). <article-title>Estimation of Globular Protein Secondary Structure from Circular Dichroism</article-title>. <source>Biochemistry</source> <volume>20</volume> (<issue>1</issue>), <fpage>33</fpage>&#x2013;<lpage>37</lpage>. <pub-id pub-id-type="doi">10.1021/bi00504a006</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.-X.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Effects of Macromolecular Crowding on the Conformational Ensembles of Disordered Proteins</article-title>. <source>J. Phys. Chem. Lett.</source> <volume>4</volume> (<issue>20</issue>), <fpage>3429</fpage>&#x2013;<lpage>3434</lpage>. <pub-id pub-id-type="doi">10.1021/jz401817x</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Quaglia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>M&#xe9;sz&#xe1;ros</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Salladini</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hatos</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pancsa</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chemes</surname>
<given-names>L. B.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>DisProt in 2022: Improved Quality and Accessibility of Protein Intrinsic Disorder Annotation</article-title>. <source>Nucleic Acids Res.</source> <volume>50</volume>, <fpage>D480</fpage>&#x2013;<lpage>D487</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab1082</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ruan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lai</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Targeting Intrinsically Disordered Proteins at the Edge of Chaos</article-title>. <source>Drug Discov. Today</source> <volume>24</volume> (<issue>1</issue>), <fpage>217</fpage>&#x2013;<lpage>227</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2018.09.017</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Simpson</surname>
<given-names>L. W.</given-names>
</name>
<name>
<surname>Good</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Leach</surname>
<given-names>J. B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Protein Folding and Assembly in Confined Environments: Implications for Protein Aggregation in Hydrogels and Tissues</article-title>. <source>Biotechnol. Adv.</source> <volume>42</volume>, <fpage>107573</fpage>. <pub-id pub-id-type="doi">10.1016/j.biotechadv.2020.107573</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sreerama</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Woody</surname>
<given-names>R. W.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Estimation of Protein Secondary Structure from Circular Dichroism Spectra: Comparison of CONTIN, SELCON, and CDSSTR Methods with an Expanded Reference Set</article-title>. <source>Anal. Biochem.</source> <volume>287</volume> (<issue>2</issue>), <fpage>252</fpage>&#x2013;<lpage>260</lpage>. <pub-id pub-id-type="doi">10.1006/abio.2000.4880</pub-id> </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Szasz</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Alexa</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Toth</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Rakacs</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Langowski</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tompa</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Protein Disorder Prevails under Crowded Conditions</article-title>. <source>Biochemistry</source> <volume>50</volume> (<issue>26</issue>), <fpage>5834</fpage>&#x2013;<lpage>5844</lpage>. <pub-id pub-id-type="doi">10.1021/bi200365j</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sz&#x151;ll&#x151;si</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>H&#xe1;zy</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Sz&#xe1;sz</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tompa</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Large Systematic Errors Compromise Quantitation of Intrinsically Unstructured Proteins</article-title>. <source>Anal. Biochem.</source> <volume>360</volume> (<issue>2</issue>), <fpage>321</fpage>&#x2013;<lpage>323</lpage>. <pub-id pub-id-type="doi">10.1016/j.ab.2006.10.027</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tunyasuvunakool</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Adler</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zielinski</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>&#x17d;&#xed;dek</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Highly Accurate Protein Structure Prediction for the Human Proteome</article-title>. <source>Nature</source> <volume>596</volume> (<issue>7873</issue>), <fpage>590</fpage>&#x2013;<lpage>596</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03828-1</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uversky</surname>
<given-names>V. N.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>A Multiparametric Approach to Studies of Self-Organization of Globular Proteins</article-title>. <source>Biochemistry (Mosc)</source> <volume>64</volume> (<issue>3</issue>), <fpage>250</fpage>&#x2013;<lpage>266</lpage>. </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uversky</surname>
<given-names>V. N.</given-names>
</name>
<name>
<surname>Fink</surname>
<given-names>A. L.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Conformational Constraints for Amyloid Fibrillation: the Importance of Being Unfolded</article-title>. <source>Biochim. Biophys. Acta (Bba) - Proteins Proteomics</source> <volume>1698</volume> (<issue>2</issue>), <fpage>131</fpage>&#x2013;<lpage>153</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbapap.2003.12.008</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uversky</surname>
<given-names>V. N.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Natively Unfolded Proteins: a point where Biology Waits for Physics</article-title>. <source>Protein Sci.</source> <volume>11</volume> (<issue>4</issue>), <fpage>739</fpage>&#x2013;<lpage>756</lpage>. <pub-id pub-id-type="doi">10.1110/ps.4210102</pub-id> </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uversky</surname>
<given-names>V. N.</given-names>
</name>
<name>
<surname>Oldfield</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Dunker</surname>
<given-names>A. K.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Intrinsically Disordered Proteins in Human Diseases: Introducing the D2 Concept</article-title>. <source>Annu. Rev. Biophys.</source> <volume>37</volume>, <fpage>215</fpage>&#x2013;<lpage>246</lpage>. <pub-id pub-id-type="doi">10.1146/annurev.biophys.37.032807.125924</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uversky</surname>
<given-names>V. N.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Protein Folding Revisited. A Polypeptide Chain at the Folding ? Misfolding ? Nonfolding Cross-Roads: Which Way to Go?</article-title> <source>Cell Mol. Life Sci. (Cmls)</source> <volume>60</volume> (<issue>9</issue>), <fpage>1852</fpage>&#x2013;<lpage>1871</lpage>. <pub-id pub-id-type="doi">10.1007/s00018-003-3096-6</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Varadi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vranken</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Guharoy</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tompa</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Computational Approaches for Inferring the Functions of Intrinsically Disordered Proteins</article-title>. <source>Front. Mol. Biosci.</source> <volume>2</volume>, <fpage>45</fpage>. <pub-id pub-id-type="doi">10.3389/fmolb.2015.00045</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ward</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Sodhi</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>McGuffin</surname>
<given-names>L. J.</given-names>
</name>
<name>
<surname>Buxton</surname>
<given-names>B. F.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>D. T.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Prediction and Functional Analysis of Native Disorder in Proteins from the Three Kingdoms of Life</article-title>. <source>J. Mol. Biol.</source> <volume>337</volume> (<issue>3</issue>), <fpage>635</fpage>&#x2013;<lpage>645</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmb.2004.02.002</pub-id> </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Whitmore</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Miles</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Mavridis</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Janes</surname>
<given-names>R. W.</given-names>
</name>
<name>
<surname>Wallace</surname>
<given-names>B. A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>PCDDB: New Developments at the Protein Circular Dichroism Data Bank</article-title>. <source>Nucleic Acids Res.</source> <volume>45</volume> (<issue>D1</issue>), <fpage>D303</fpage>&#x2013;<lpage>D307</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw796</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Woody</surname>
<given-names>R. W.</given-names>
</name>
<name>
<surname>Berova</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2000</year>). <source>Circular Dichroism: Principles and Applications</source>. <publisher-name>Wiley VCH</publisher-name>. </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Woollett</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Whitmore</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Janes</surname>
<given-names>R. W.</given-names>
</name>
<name>
<surname>Wallace</surname>
<given-names>B. A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>ValiDichro: a Website for Validating and Quality Control of Protein Circular Dichroism Spectra</article-title>. <source>Nucleic Acids Res.</source> <volume>41</volume> (<issue>Web Server issue</issue>), <fpage>W417</fpage>&#x2013;<lpage>W421</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt287</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>