<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mol. Biosci.</journal-id>
<journal-title>Frontiers in Molecular Biosciences</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mol. Biosci.</abbrev-journal-title>
<issn pub-type="epub">2296-889X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1244029</article-id>
<article-id pub-id-type="doi">10.3389/fmolb.2023.1244029</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Molecular Biosciences</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Chemical shift transfer: an effective strategy for protein NMR assignment with ARTINA</article-title>
<alt-title alt-title-type="left-running-head">Wetton et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fmolb.2023.1244029">10.3389/fmolb.2023.1244029</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wetton</surname>
<given-names>Henry</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2353979/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Klukowski</surname>
<given-names>Piotr</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2419935/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Riek</surname>
<given-names>Roland</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/141208/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>G&#xfc;ntert</surname>
<given-names>Peter</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/765658/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Institute of Molecular Physical Science</institution>, <institution>ETH Zurich</institution>, <addr-line>Zurich</addr-line>, <country>Switzerland</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Institute of Biophysical Chemistry</institution>, <institution>Goethe University Frankfurt</institution>, <addr-line>Frankfurt</addr-line>, <country>Germany</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Chemistry</institution>, <institution>Tokyo Metropolitan University</institution>, <addr-line>Hachioji</addr-line>, <country>Japan</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1127101/overview">Caterina Alfano</ext-link>, Ri.MED Foundation, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/464865/overview">Georg Kontaxis</ext-link>, University of Vienna, Austria</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1591129/overview">Rafal Augustyniak</ext-link>, University of Warsaw, Poland</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Piotr Klukowski, <email>piotr.klukowski@phys.chem.ethz.ch</email>; Roland Riek, <email>roland.riek@phys.chem.ethz.ch</email>; Peter G&#xfc;ntert, <email>peter.guentert@phys.chem.ethz.ch</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>10</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>10</volume>
<elocation-id>1244029</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>06</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>09</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Wetton, Klukowski, Riek and G&#xfc;ntert.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Wetton, Klukowski, Riek and G&#xfc;ntert</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Chemical shift transfer (CST) is a well-established technique in NMR spectroscopy that utilizes the chemical shift assignment of one protein (source) to identify chemical shifts of another (target). Given similarity between source and target systems (e.g., using homologs), CST allows the chemical shifts of the target system to be assigned using a limited amount of experimental data. In this study, we propose a deep-learning based workflow, ARTINA-CST, that automates this procedure, allowing CST to be carried out within minutes or hours of computational time and strictly without any human supervision. We characterize the efficacy of our method using three distinct synthetic and experimental datasets, demonstrating its effectiveness and robustness even when substantial differences exist between the source and target proteins. With its potential applications spanning a wide range of NMR projects, including drug discovery and protein interaction studies, ARTINA-CST is anticipated to be a valuable method that facilitates research in the field.</p>
</abstract>
<kwd-group>
<kwd>NMR</kwd>
<kwd>machine learning</kwd>
<kwd>automated spectra analysis</kwd>
<kwd>automated assignment</kwd>
<kwd>ARTINA</kwd>
<kwd>FLYA</kwd>
<kwd>protein</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Structural Biology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Recently, ARTINA (<xref ref-type="bibr" rid="B11">Klukowski et al., 2022</xref>), the first workflow that automates the analysis of protein NMR data for signal identification, resonance assignment (<xref ref-type="bibr" rid="B16">Schmidt and G&#xfc;ntert, 2012</xref>), and structure determination (<xref ref-type="bibr" rid="B7">G&#xfc;ntert et al., 1997</xref>; <xref ref-type="bibr" rid="B6">G&#xfc;ntert and Buchner, 2015</xref>) was developed, demonstrating the capability of machine learning to advance biomolecular NMR. With ARTINA, the entire NMR data analysis process can be completed on a web server (<xref ref-type="bibr" rid="B10">Klukowski et al., 2023</xref>) without human supervision and within hours after the NMR measurements are completed, replacing weeks or months of human labor. The procedure requires spectra with an information content that is sufficient to unambiguously assign chemical shifts without any prior knowledge about the protein structure. The exact amount of experimental data required for <italic>de novo</italic> chemical shift assignment with ARTINA depends on the signals&#x2019; resolution and signal-to-noise ratio. Typically, the requirement for the measurement time on expensive NMR spectrometers scales to approximately one or two weeks.</p>
<p>Often, one would like to study a system by NMR that is similar to an already assigned protein (<xref ref-type="bibr" rid="B17">Thompson et al., 2012</xref>). Examples of this include homologous proteins (<xref ref-type="bibr" rid="B15">Redfield and Robertson, 1991</xref>; <xref ref-type="bibr" rid="B4">Bartels et al., 1996</xref>), proteins studied under different experimental conditions (e.g., temperature, pH value, ligand concentration) (<xref ref-type="bibr" rid="B9">Jang et al., 2012</xref>; <xref ref-type="bibr" rid="B3">Banelli et al., 2017</xref>; <xref ref-type="bibr" rid="B19">Zieba et al., 2018</xref>), or proteins with bound ligands and in apo form (<xref ref-type="bibr" rid="B13">Orts and Gossert, 2018</xref>; <xref ref-type="bibr" rid="B12">Laveglia et al., 2021</xref>; <xref ref-type="bibr" rid="B14">Plata et al., 2023</xref>). In such cases, it is desirable to use knowledge about the known system, in particular its assigned chemical shifts, as complementary input for the automated analysis of a related target system with ARTINA. Such complementary input typically makes it possible to assign chemical shifts or determine the protein structure using a smaller set of spectra. In this work we evaluate the accuracy of ARTINA-CST and develop guidelines to utilize our method for chemical shift transfer applications.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 Chemical shift transfer</title>
<p>The interface required for chemical shift transfer (CST) is included within the FLYA algorithm used by ARTINA for chemical shift assignment (<xref ref-type="bibr" rid="B16">Schmidt and G&#xfc;ntert, 2012</xref>). The function of this algorithm has been described in detail by Schmidt and G&#xfc;ntert; in brief, a list of expected peaks is constructed from the protein of interest&#x2019;s amino acid sequence, then this list is mapped to a list of measured peaks generated by manual or automated peak picking. The mapping is improved iteratively using global and local optimization methods (<xref ref-type="bibr" rid="B5">Bartels et al., 1997</xref>). This procedure is repeated for a series of replicates and the final output is determined as a majority consensus of these.</p>
<p>In order to initialize the assignment of expected peaks to measured peaks, the default procedure within FLYA uses statistics from the Biological Magnetic Resonance Data Bank (BMRB) (<xref ref-type="bibr" rid="B8">Hoch et al., 2023</xref>) to construct an initial &#x201c;search space&#x201d; for each expected peak, from which a measured peak is picked at random to assign it (<xref ref-type="bibr" rid="B16">Schmidt and G&#xfc;ntert, 2012</xref>). For each atom in a protein, this search space is defined by a statistical distribution parametrized by the mean and standard deviation of chemical shifts of the given atom type over all occurrences of its amino acid type within the BMRB. Due to the scale of the database and the high number of peaks in protein NMR spectra, this usually generates a large area containing also many incorrect possibilities for assignment, the center of which can deviate from the peak&#x2019;s true position. By improving the initial search space for each atom, i.e., reducing its size and shifting its center closer to the true position, the accuracy of FLYA assignments can therefore be improved significantly (<xref ref-type="bibr" rid="B4">Bartels et al., 1996</xref>; <xref ref-type="bibr" rid="B1">Aeschbacher et al., 2013</xref>). The essence of the CST procedure relies on this concept, aiming to provide a better estimate of the peak&#x2019;s true position and therefore also allowing for a smaller search space size (<xref ref-type="fig" rid="F1">Figure 1A</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Graphical abstract of the ARTINA-CST workflow. <bold>(A)</bold> CST affects automated chemical shift assignment through the selection of peak search spaces (light red rectangle). Without CST, the search space is estimated based on statistical information extracted from the BMRB database, whereas with CST, it is determined from a previously assigned source protein (the dark red circle indicates the chemical shifts of the source protein, and the highlighted area is specified by a user-defined parameter). The resulting area is centered closer to the target peak of interest (dark green oval) and contains less incorrect peaks (light green ovals). <bold>(B)</bold> The chemical shifts of the source protein are prepared by sequence alignment to the target protein and subsequent extraction of search spaces from the chemical shifts of the aligned residues. Measured NMR data is then input to FLYA as a list of peaks extracted automatically from spectra, yielding the assigned shifts of the target protein.</p>
</caption>
<graphic xlink:href="fmolb-10-1244029-g001.tif"/>
</fig>
<p>CST is performed as follows (<xref ref-type="fig" rid="F1">Figure 1B</xref>): Given a source protein with a known assignment of chemical shifts and a target protein whose shifts have yet to be assigned, first a mapping of source atoms to target atoms must be defined. In the trivial case where source and target protein are identical in sequence, each source atom can simply be mapped to its copy in the target. However, this need not always be the case, and transfer of assignment data between different protein samples could also be highly beneficial in many cases. In this work, we compared pairs of distinct proteins by first performing a local pairwise sequence alignment tuned to produce large aligned sub-sequences between a small number of wide gaps (<xref ref-type="sec" rid="s10">Supplementary Table S1</xref>). Each identical pair of amino acids in this alignment could then be used for CST. Aligned pairs of different amino acids and amino acids aligned to gaps were simply disregarded, falling back to FLYA&#x2019;s default setting.</p>
<p>Based on this source-target mapping, new initial search spaces can be defined for each atom in the target protein, centered around the corresponding chemical shift in the source. Since this new center is likely to be more accurate, i.e., closer to the true position of the atom&#x2019;s shift than the mean value obtained from BMRB, the search space&#x2019;s width can then be reduced (higher precision), further facilitating its assignment.</p>
<p>The determined search space centers and sizes are then entered into FLYA as a list of mean values and standard deviations (&#x201c;chemical shift statistics&#x201d;), with the algorithm automatically applying the default BMRB-based search space for any atom not included.</p>
<p>When using FLYA as part of the ARTINA pipeline, automated visual analysis of each spectrum generates the experimental peaks to be used as input (<xref ref-type="bibr" rid="B11">Klukowski et al., 2022</xref>). For this work, execution was stopped after obtaining an assignment from FLYA. In practice, however, the output assignments can also be passed to the next stages of ARTINA, e.g., structure determination.</p>
</sec>
<sec id="s2-2">
<title>2.2 NMR data</title>
<p>The dataset established for the training and testing of ARTINA models (<xref ref-type="bibr" rid="B11">Klukowski et al., 2022</xref>) was the source of raw NMR spectral data in this study. Since CST is most efficient if only a small number of spectra have to be measured for the target protein, we used a small subset of NMR spectra types from the ARTINA dataset, consisting only of the 2D HSQC ([<sup>1</sup>H, <sup>13</sup>C]-HSQC, [<sup>1</sup>H, <sup>15</sup>N]-HSQC) and 3D NOESY (<sup>13</sup>C-resolved [<sup>1</sup>H, <sup>1</sup>H] NOESY, <sup>15</sup>N-resolved [<sup>1</sup>H, <sup>1</sup>H] NOESY) spectra. Unless specified otherwise, we used this small subset of spectra for a representative set of 15 proteins for all experiments in this work (<xref ref-type="sec" rid="s10">Supplementary Table S2</xref>).</p>
</sec>
<sec id="s2-3">
<title>2.3 Test examples for automated chemical shift transfer</title>
<sec id="s2-3-1">
<title>2.3.1 Random perturbation</title>
<p>To evaluate the capabilities of ARTINA-CST, we initially generated CST test examples from the NMR data specified in <xref ref-type="sec" rid="s2-2">section 2.2</xref>. Each test example was composed of the chemical shift list of the target (typically BMRB deposition) and source (randomly perturbed BMRB deposition) proteins, complemented by the set of NMR spectra of the target. The procedure ensured that each target protein had a corresponding source from which to draw chemical shift information, while also enabling quantification of the difference between the source and target shifts. The source chemical shift lists were created by manually perturbing deposited chemical shifts with Gaussian additive noise.</p>
<p>The standard deviation <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the normal distribution from which the perturbations were drawn was chosen individually for each atom <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, using the standard deviation of the chemical shift over all occurrences of the same atom and amino acid type in the BMRB (<inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>BMRB</mml:mtext>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>):<disp-formula id="equ1">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>c</mml:mi>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>BMRB</mml:mtext>
</mml:msubsup>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a scaling constant (<inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mn>0.2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1.0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>) selected to validate different experimental settings.</p>
<p>When applying this data for CST, the uncertainty value used to determine the search space width was set to the larger of either:<list list-type="simple">
<list-item>
<p>&#x2022; the standard deviation used for perturbation (<inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>), or</p>
</list-item>
<list-item>
<p>&#x2022; a small tolerance of 0.04&#xa0;ppm for <sup>1</sup>H nuclei, and 0.4&#xa0;ppm for both <sup>13</sup>C and <sup>15</sup>N.</p>
</list-item>
</list>
</p>
<p>To generate test examples, perturbations were applied to a specific fraction <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mn>0.2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1.0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> of the total number of chemical shifts within each protein, selected at random.</p>
<p>By evaluating all combinations of the parameters <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> specified above, we conducted a total of nine distinct experiments. Each entailed a full assignment (backbone and sidechain) of the 15 benchmark proteins, resulting in a total of 135 assignments reported in the experimental section.</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Structure-based perturbation</title>
<p>The subsequent series of experiments incorporated protein structure information into the source shift list generation procedure. The routine was designed to replicate an experimental setting where chemical shifts change between source and target due to such factors as point mutations or ligand binding at specific sites. We employed a random approach like the one outlined in the previous section. However, the standard deviation of the perturbation was defined as a function of the distance from the &#x201c;perturbation centers&#x201d; (<inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>)&#x2014;amino acids selected from the sequence of the target protein at random. The source chemical shift list generation procedure was the following:<list list-type="simple">
<list-item>
<p>1. For each perturbation center <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and each atom <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in the protein structure <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, calculate the distance (<inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) between the backbone amide nitrogen atoms of the perturbation center <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the amino acid containing atom <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>2. Calculate the decay factor <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> for each atom <inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which combines the impact of all perturbation centers:</p>
</list-item>
</list>
<disp-formula id="equ2">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where<disp-formula id="equ3">
<mml:math id="m21">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mtext>&#x2009;if&#x2009;</mml:mtext>
<mml:mi>x</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mtext>&#x2009;if&#x2009;</mml:mtext>
<mml:mi>x</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>is the step function parametrized by a distance cutoff <inline-formula id="inf19">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.<list list-type="simple">
<list-item>
<p>3. Normalize each decay factor: <inline-formula id="inf20">
<mml:math id="m23">
<mml:mrow>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>/</mml:mo>
<mml:munder>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to ensure that <inline-formula id="inf21">
<mml:math id="m24">
<mml:mrow>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is a scalar value in the range <inline-formula id="inf22">
<mml:math id="m25">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>4. Calculate the standard deviation (<inline-formula id="inf23">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) of the additive noise for the <inline-formula id="inf24">
<mml:math id="m27">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> -th atom: <inline-formula id="inf25">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mi>c</mml:mi>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>BMRB</mml:mtext>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf26">
<mml:math id="m29">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mn>0.2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1.0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a constant used to validate different experimental settings.</p>
</list-item>
<list-item>
<p>5. Set the uncertainty value used to determine the search space width for the atom <inline-formula id="inf27">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to the larger of either:</p>
</list-item>
<list-item>
<p>&#x2022; the standard deviation used for perturbation (<inline-formula id="inf28">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>), or</p>
</list-item>
<list-item>
<p>&#x2022; a small tolerance of 0.03&#xa0;ppm for <sup>1</sup>H nuclei, and 0.4&#xa0;ppm for both <sup>13</sup>C and <sup>15</sup>N.</p>
</list-item>
</list>
</p>
<p>These steps form an alternative method to define the atom-specific chemical shift noise distribution <inline-formula id="inf29">
<mml:math id="m32">
<mml:mrow>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, contrasting the fully randomized setting described in the preceding section. Aside from this distinction, both experimental settings follow the same logic. An example rendering of <inline-formula id="inf30">
<mml:math id="m33">
<mml:mrow>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> for three perturbation centers is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Three-dimensional rendering of decay factors (<inline-formula id="inf31">
<mml:math id="m34">
<mml:mrow>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) using the protein PDB 1SE9 as an example. Three perturbation centers are represented by blue spheres that are overlaid with a light ribbon representation of the protein backbone in cyan. Decay factor values are presented for each residue as stick colors on a scale from 0 (white) to 1 (dark red). The resulting assignment accuracy is shown for each heavy atom as spheres, with assignment errors up to 0.4&#xa0;ppm in dark green color. Incorrectly assigned heavy atoms are not present in this assignment result. For atoms with no visible sphere, automated assignments were made, however no reference manual shift was available for error calculation.</p>
</caption>
<graphic xlink:href="fmolb-10-1244029-g002.tif"/>
</fig>
<p>In the experiment, we used the 15 benchmark proteins to generate 900 randomized CST test examples by drawing <inline-formula id="inf32">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from a uniform distribution over the interval 5&#x2013;25&#xa0;&#xc5;, the number of perturbation centers from a categorical distribution over the integers 1&#x2013;10, and positions of perturbation centers by selecting the specified number of protein residues from the sequence at random.</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Homologous proteins</title>
<p>To evaluate CST with experimental data, we investigated sequence homology between proteins in the ARTINA dataset (<xref ref-type="bibr" rid="B11">Klukowski et al., 2022</xref>) (<xref ref-type="sec" rid="s10">Supplementary Table S2</xref>) and both RefDB (<xref ref-type="bibr" rid="B18">Zhang et al., 2003</xref>) and BMRB using sequence alignment parameters specified in <xref ref-type="sec" rid="s10">Supplementary Table S1</xref>. As sequence alignment scores had no upper bound, they were normalized by the score of aligning each protein to itself. If the homolog protein was found in both RefDB and BMRB, preference was given to RefDB. We refer to the protein from the ARTINA dataset as the target and from RefDB/BMRB as the source protein.</p>
<p>All pairs with normalized sequence alignment score above 80% were selected for chemical shift alignment. In this step we propagated information from the sequence alignment into the source chemical shift list. This was indispensable as differences between the source and target sequences, such as insertion or deletion, require appropriate reindexing of chemical shifts in the source before applying CST to the target.</p>
<p>Subsequently, each shift in the aligned source list was compared with the corresponding target shift. An aligned shift was considered &#x201c;correct&#x201d; if it was within a given tolerance from the target position (0.03&#xa0;ppm for <sup>1</sup>H, 0.4&#xa0;ppm for <sup>13</sup>C/<sup>15</sup>N). Then, we calculated the &#x201c;fraction of correct aligned shifts&#x201d; for each pair of source/target homologous proteins, which is defined as the ratio of correct aligned shifts to total shifts in the aligned list.</p>
<p>In this experiment, we used protein pairs with a fraction of correct aligned shifts greater than or equal to 50%. Combining with the requirement of &#x3e;80% sequence alignment score, we identified 12 source-target pairs for which experimental spectra were available in the ARTINA benchmark. These pairs corresponded to 9 distinct target proteins (<xref ref-type="sec" rid="s10">Supplementary Figure S2</xref>, <xref ref-type="sec" rid="s10">Supplementary Table S3</xref>).</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Reference experiments</title>
<p>To investigate the impact of complementary input (chemical shift list of the source protein) on the accuracy of the chemical shift assignment of the target protein, we first evaluated the baseline performance of <italic>de novo</italic> automated assignments of target proteins, without any transfer information (<xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>).</p>
<p>Subsequently, we ran the chemical shift transfer in an idealized setting, where each target protein&#x2019;s own chemical shift list was passed as the transfer source. This corresponds to the scenario, unlikely in practice, in which the chemical shifts do not change between the source and target proteins, or that the input shift list for CST is equal to the expected output. It is important to note that FLYA still uses peak lists prepared for a minimal set of experimental spectra ([<sup>1</sup>H, <sup>13</sup>C]-HSQC, [<sup>1</sup>H, <sup>15</sup>N]-HSQC, <sup>13</sup>C-resolved [<sup>1</sup>H, <sup>1</sup>H] NOESY, <sup>15</sup>N-resolved [<sup>1</sup>H, <sup>1</sup>H] NOESY) to perform chemical shift transfer. Therefore, it cannot simply copy the input shifts to the output, and some chemical shifts cannot be assigned due to missing signals in the input data. However, this experimental setting represents the best possible initial search space positions for FLYA, and therefore can be used to estimate an upper bound for CST accuracy. In this case, the width of each search space was set to 0.04&#xa0;ppm for <sup>1</sup>H nuclei and 0.4&#xa0;ppm for <sup>13</sup>C and <sup>15</sup>N nuclei.</p>
<p>As expected, we observed a significant improvement in chemical shift accuracy by 23.5 percentage points (pp) in the idealized reference case, as compared to <italic>de novo</italic> assignment, where no complementary input was used. The average assignment accuracy of 68.8% for 15 proteins used in the <italic>de novo</italic> experiment increased to 92.3% using CST (<xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>). The fact that FLYA was unable to attain 100% accuracy in these reference experiments was likely mainly due to the minimal set of spectra used (<xref ref-type="sec" rid="s10">Supplementary Table S2</xref>).</p>
<p>In both reference experiments (<xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>) the assignment errors had a tendency to accumulate in side-chains (90.6% accuracy with CST, 66.3% without), leaving the accuracy of the backbone assignment above the average (95.3% accuracy with CST, 73.4% without)&#x2014;a result consistent with previous studies of FLYA (<xref ref-type="bibr" rid="B16">Schmidt and G&#xfc;ntert, 2012</xref>) and ARTINA (<xref ref-type="bibr" rid="B11">Klukowski et al., 2022</xref>).</p>
<p>Another notable observation is that the availability of source chemical shifts affects the variance of the output accuracy. In <italic>de novo</italic> experiments, the discrepancy between the most and least accurately assigned proteins is 34.9 pp, compared to only 13.8 pp in the idealized CST case.</p>
<p>Overall, the experiments we have conducted, ranging from scenarios with minimal to maximal information available for chemical shift transfer, indicate that source protein information provides useful guidance for the combinatorial optimization that reduces the ambiguity of the assignment, resulting in fewer errors and enhancing the overall reliability of protein assignment.</p>
</sec>
<sec id="s3-2">
<title>3.2 The impact of random and structure-based perturbations</title>
<p>To characterize ARTINA-CST under more realistic conditions, we carried out over 1,000 automated chemical shift transfers with 15 proteins using the test examples described in <xref ref-type="sec" rid="s2-3-1">sections 2.3.1</xref> and <xref ref-type="sec" rid="s2-3-2">2.3.2</xref>.</p>
<p>In the first series of experiments, a variable fraction of atoms was selected at random for perturbation (20%, 50% and 100%), as described in <xref ref-type="sec" rid="s2-3-1">section 2.3.1</xref>, imitating a variable degree of discrepancies between known source chemical shifts and the target protein. Perturbed chemical shifts were used as input for the ARTINA-CST procedure together with 4 NMR spectra of the target protein ([<sup>1</sup>H, <sup>13</sup>C]-HSQC, [<sup>1</sup>H, <sup>15</sup>N]-HSQC, <sup>13</sup>C-resolved [<sup>1</sup>H, <sup>1</sup>H] NOESY, <sup>15</sup>N-resolved [<sup>1</sup>H, <sup>1</sup>H] NOESY). ARTINA-CST automatically extracted cross-peak positions from experimental data and combined them with the corresponding perturbed shift list, yielding the assignment of the target protein. This result was compared with <italic>de novo</italic> ARTINA assignment, which involved the same procedure, but without the perturbed chemical shift list as input.</p>
<p>The use of CST turned out to be highly beneficial in all three experimental settings (20%, 50% and 100% shift perturbation) and for almost all proteins included in the study. The relative improvement in assignment accuracy is depicted in <xref ref-type="fig" rid="F3">Figures 3A&#x2013;C</xref> by the ratio between the accuracy of the chemical shift assignment with the CST procedure and with the <italic>de novo</italic> approach, with a value of 1.0 corresponding to a neutral effect of CST.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Improvement of chemical shift assignment with CST using randomly and structure-based perturbed shifts. On the vertical axis, the change in accuracy is given as the ratio between the chemical shift assignment accuracy with CST and corresponding <italic>de novo</italic> assignments without information about the source protein. On the horizontal axis, the fraction of shifts perturbed refers to all shifts where perturbations were applied, regardless of the size of the perturbation. The trend of each plot is modelled using Gaussian process regression and shown in blue, with a 95% confidence interval shown in orange. &#x201c;Scaling coefficient&#x201d; refers to the constant <inline-formula id="inf33">
<mml:math id="m36">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> used to scale the standard deviation of the applied perturbations (see <xref ref-type="sec" rid="s2-3-1">sections 2.3.1</xref> and <xref ref-type="sec" rid="s2-3-2">2.3.2</xref>). <bold>(A&#x2013;C)</bold> Presents the results of the experiments with random and <bold>(D&#x2013;F)</bold> with structure-based perturbations.</p>
</caption>
<graphic xlink:href="fmolb-10-1244029-g003.tif"/>
</fig>
<p>As expected, the value of the ratio depends on the parameters <inline-formula id="inf34">
<mml:math id="m37">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf35">
<mml:math id="m38">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the test generation procedure (<xref ref-type="sec" rid="s2-3-1">section 2.3.1</xref>). In the least challenging setting (<xref ref-type="fig" rid="F3">Figure 3A</xref>), the scaling factor (<inline-formula id="inf36">
<mml:math id="m39">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>) largely restricts chemical shift deviations between source and target chemical shift list. As the chemical shift values in the source and target shift lists lie in close proximity, even for perturbed atoms, the CST transfer procedure yields similar performance regardless of the number of atoms perturbed, demonstrating a 1.36-fold relative improvement on average as compared to the <italic>de novo</italic> runs.</p>
<p>When the scaling coefficient was set to a moderately higher value (<inline-formula id="inf37">
<mml:math id="m40">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>), the effect of the larger of chemical shift perturbations on the overall CST accuracy was apparent (<xref ref-type="fig" rid="F3">Figure 3B</xref>), with the relative improvement averaged over all proteins varying between 1.37 and 1.29 for 0% and 100% perturbed chemical shifts respectively. Finally, the strongest perturbation (<inline-formula id="inf38">
<mml:math id="m41">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <xref ref-type="fig" rid="F3">Figure 3C</xref>) result in a further decrease in CST accuracy, as compared with previous settings (<xref ref-type="fig" rid="F3">Figures 3A, B</xref>), but preserving strong positive impact (1.37-fold&#x2013;1.23-fold improvement) relative to <italic>de novo</italic> calculations.</p>
<p>Subsequently, we repeated the above experiment using structure-based perturbations (<xref ref-type="sec" rid="s2-3-2">section 2.3.2</xref>) instead of random ones. The results for these assignments show a strong resemblance to those for the fully randomized perturbations (<xref ref-type="fig" rid="F3">Figures 3D&#x2013;F</xref>), including correlation between CST accuracy and <inline-formula id="inf39">
<mml:math id="m42">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf40">
<mml:math id="m43">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> parameters of the test example preparation procedure. The points with 0% perturbations correspond to the idealized case, as described in <xref ref-type="sec" rid="s3-1">section 3.1</xref>.</p>
<p>Overall, we identified no major changes in the performance of the algorithm that depend on the spatial distribution of the chemical shift perturbation, and in both experimental settings the number of perturbations and corresponding variance were the primary factors affecting the accuracy of the procedure.</p>
</sec>
<sec id="s3-3">
<title>3.3 Case study with the lipoprotein Spr NlpC/P60 domain</title>
<p>For individual cases, we observed a particularly large improvement in the assignment obtained with the CST procedure. For instance, the C-terminal NlpC/P60 domain of lipoprotein Spr from <italic>Escherichia coli</italic> [PDB 2K1G (<xref ref-type="bibr" rid="B2">Aramini et al., 2008</xref>)], showed poor performance in <italic>de novo</italic> chemical shift assignment (48.3% accuracy), compared to 94.7% obtained in the idealized case of CST (<xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
<p>Even in the experimental setting with the strongest perturbation (standard deviation equal to BMRB standard deviation and perturbations applied to all shifts, which resulted in the large majority of shifts being moved significantly from their original positions), the fraction of correct assignments for this protein was raised by 20.4 pp upon application of the CST procedure. In all other cases the positive impact of the chemical shift transfer was even stronger (30.8&#x2013;49.8 pp improvement, depending on the experimental setting).</p>
<p>The primary reason for such improvements was the ability of the CST method to resolve chemical shifts in the proximity of the dynamic loop of residues 16&#x2013;32, 79&#x2013;81, 90&#x2013;93, 99&#x2013;101, long positively charged side chains (37.3% without CST vs. 91.5% with idealized CST) and aromatics (54.9% vs. 93.7%).</p>
</sec>
<sec id="s3-4">
<title>3.4 Chemical shift transfer with homologous proteins</title>
<p>As described in <xref ref-type="sec" rid="s2-4">section 2.4</xref>, we used pairs of homologous proteins identified in RefDB/BMRB and the ARTINA benchmark of NMR spectra to assess the performance of the chemical shift transfer in fully experimental setting. In this procedure, the chemical shift list deposited in RefDB/BMRB was regarded as source, and four NMR spectra ([<sup>1</sup>H, <sup>13</sup>C]-HSQC, [<sup>1</sup>H, <sup>15</sup>N]-HSQC, <sup>13</sup>C-resolved [<sup>1</sup>H, <sup>1</sup>H] NOESY, <sup>15</sup>N-resolved [<sup>1</sup>H, <sup>1</sup>H] NOESY) were used by ARTINA-CST as input for the transfer procedure to the target system. In this experiment all cross-peaks in the abovementioned spectra were identified automatically by deep neural network models included in the ARTINA visual spectrum analysis layer. As in the previous experiments, we performed for each target protein additional <italic>de novo</italic> assignments to assess the relative performance of the CST procedure.</p>
<p>The results show an overall mean improvement by CST over <italic>de novo</italic> assignment of 7.4% (0.0%&#x2013;14.4%) in the chemical shift assignment accuracy of all shifts and 9.7% (1.2%&#x2013;23.9%) for the backbone NH groups (<xref ref-type="fig" rid="F4">Figure 4</xref>). Out of 12 homolog pairs, the impact of CST was positive both for all shifts and NH groups in 11 cases. Only in one case (2JVD) the impact of CST was neutral for all shifts and positive for NH groups.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Comparison of the accuracy of the chemical shift assignment with and without CST. In the experiment 12 pairs of 9 homologous proteins were used. Each target protein is represented in a distinct color (for 2LEA, 2LL8 and 2LRH, two source proteins from BMRB were identified). Circles represent assignment accuracy over all atom types, whereas crosses signify the accuracy in the same experiments evaluated for the backbone amide N/H atoms only.</p>
</caption>
<graphic xlink:href="fmolb-10-1244029-g004.tif"/>
</fig>
<p>ARTINA-CST has a single parameter, <inline-formula id="inf41">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>CST</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which corresponds to the level of shift perturbation one expects to observe when transferring chemical shift assignments from the source to the target protein. It plays the same role as the scaling factor <inline-formula id="inf42">
<mml:math id="m45">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> for the synthetic data preparation in <xref ref-type="sec" rid="s2-3-1">section 2.3.1</xref>, indicating the fraction of BMRB variance in individual shifts we expect to observe in a particular dataset. In this experiment, <inline-formula id="inf43">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>CST</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was set to 1.0, indicating lack of a prior assumption about the variance of the distribution (i.e., we expect the chemical shifts in the target protein to deviate from the source the same way the individual shifts deposited in the BMRB database deviate from their mean value). Despite this conservative assumption, ARTINA-CST still yielded a substantial improvement in the chemical shift assignment accuracy, as compared with <italic>de novo</italic> assignment. Specific tuning of the expected perturbation parameter is expected to result in further improvement of the ARTINA-CST performance, as it provides weak constraints on the initial search-space for individual chemical shifts.</p>
<p>In a second series of experiments, we therefore carried out chemical shift transfers between the 12 homolog pairs using different values of the expected perturbations, <inline-formula id="inf44">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>CST</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.10, 0.15, 0.20, 0.25, 0.50, 1.00, as well as with fixed-size search spaces of 0.09&#xa0;ppm for <sup>1</sup>H and 1.2&#xa0;ppm for <sup>13</sup>C/<sup>15</sup>N shifts that are independent of BMRB shift distributions. The results indicate that decreasing the size of the initial search space proved effective to increase ARTINA-CST accuracy (<xref ref-type="fig" rid="F5">Figure 5</xref>). For each automated chemical shift assignment in this experiment, we calculated the ratio between the accuracies of CST-based and <italic>de novo</italic> assignment. With the lack of prior assumptions about the variance of chemical shifts (<inline-formula id="inf45">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>CST</mml:mtext>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>), ARTINA-CST achieved 10% median improvement over <italic>de novo</italic> assignment, for which no information from the homologous protein was used. As the value of the expected perturbation parameter <inline-formula id="inf46">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>CST</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> decreased, the overall performance of the method increased and saturated at about 26% for <inline-formula id="inf47">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>CST</mml:mtext>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mn>0.1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.25</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. As two reference experiments, we evaluated the accuracy of chemical shift transfer with fixed tolerances (green box, <xref ref-type="fig" rid="F5">Figure 5</xref>) and with an idealistic optimal reference, where the information about target chemical shifts is assumed to be known (upper bound accuracy) (blue box, <xref ref-type="fig" rid="F5">Figure 5</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Impact of search space size on CST accuracy. The change in accuracy is given as the ratio between the accuracies with and without chemical shift transfer. The left-hand side represents assignment accuracy over all atom types, whereas the right-hand side signifies the accuracy in the same experiments only evaluated over all amide N/H atoms. Within each set of experiments, the furthest-right case (blue) represents the &#x201c;ideal&#x201d; reference experiment, using the target protein shift list as source for chemical shift transfer. The neighboring case (green) used a fixed-size search space of 0.09&#xa0;ppm for <sup>1</sup>H and 1.2&#xa0;ppm for <sup>13</sup>C/<sup>15</sup>N. The other boxes (red) represent a variable initial search space size calculated as a fraction of the BMRB standard deviation of chemical shifts.</p>
</caption>
<graphic xlink:href="fmolb-10-1244029-g005.tif"/>
</fig>
<p>Chemical shift transfer impacts full assignment and amide N/H assignment in similar way (<xref ref-type="fig" rid="F5">Figure 5</xref>). For <inline-formula id="inf48">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>CST</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 1.0, the relative median increase in assignment accuracy is 11%. The quality of the solution increases with decreasing expected perturbation coefficient, saturating at 20% for <inline-formula id="inf49">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mtext>CST</mml:mtext>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mn>0.1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.25</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The relative improvement for amide groups is smaller compared to all shifts, because the generally higher accuracy of NH shift assignments in <italic>de novo</italic> experiments leaves smaller room for improvements (68.0% and 77.2% accuracy for all shifts and NH groups respectively).</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>An assigned set of chemical shifts establishes a basis for various studies in protein NMR spectroscopy. It facilitates structure elucidation, with chemical shifts being instrumental in determining hydrogen spatial contacts and providing insight into the three-dimensional architecture of the macromolecule. Beyond elucidating static structures, chemical shift assignments offer a unique window into protein dynamics, allowing for the monitoring of temporal changes, facilitating the observation of protein folding processes, conformational alterations, and molecular interactions. In the field of protein-ligand studies, the chemical shifts typically exhibit changes upon ligand binding, thereby pinpointing the site of interaction, as well as details regarding its molecular mechanism and affinity.</p>
<p>In this study we focused on chemical shift transfer&#x2013;a technique that allows to find chemical shift assignment of a target protein of interest, given a (small) set of experimental spectra and the shift assignment of a similar source protein (e.g., a homolog). The technique is particularly suitable for studies of protein interactions, such as protein-ligand complexes, where information about the structure in apo form can be utilized to model the structure upon binding. Other applications include studies of protein mutations, where a wild-type structure with its assignment can be used as a source of the transfer for a series of mutants. Finally, chemical shift transfer finds its applications in studies of proteins under different physical conditions. In all experimental settings evaluated in this study, the goal of the chemical shift transfer was to find chemical shifts of the target protein with a small amount of experimental data, thereby reducing the measurement time from about 1 to 2&#xa0;weeks to 2&#x2013;3 days ([<sup>1</sup>H,<sup>13</sup>C]-HSQC, [<sup>1</sup>H,<sup>15</sup>N]-HSQC, and combined <sup>15</sup>N,<sup>13</sup>C-resolved [<sup>1</sup>H,<sup>1</sup>H]-NOESY).</p>
<p>In this work we built upon our previous work with ARTINA (<xref ref-type="bibr" rid="B11">Klukowski et al., 2022</xref>) and FLYA (<xref ref-type="bibr" rid="B16">Schmidt and G&#xfc;ntert, 2012</xref>) to establish a fully automated workflow that performs chemical shift transfer automatically, strictly without any human involvement. Subsequently, we carried out over 1,000 automated chemical shift assignments to demonstrate the performance of ARTINA-CST approach and characterize its properties under different experimental settings.</p>
<p>We demonstrated the boundary performance of ARTINA-CST by carrying out automated CST with complete information about the target system and without any information from the homolog structure. Subsequently, we characterized the performance of our method, depending on such factors as the similarity of source and target protein chemical shifts, the variance of chemical shift perturbations, or the spatial distribution of chemical shift deviations. Finally, we demonstrated the performance of our approach using pairs of homolog proteins extracted from RefDB/BMRB databases, which have experimental data available in the ARTINA dataset. In all these experiments, we demonstrated the benefits of CST for the automated assignment of protein NMR spectra whenever appropriate data is available. Even in presence of large differences between the target protein and the source chemical shifts, CST allows an effective transfer of information to improve the assignment while requiring only a small set of spectra for the target protein. This experiment was carried out with a minimal set of NMR spectra for source-target pairs with at least 80% sequence homology (and correspondingly lower sequence identity). We expect CST to be possible also at even lower sequence similarity, where, however, it might be necessary to compensate for the larger number and size of the chemical shift differences by measuring one or more additional spectra (for instance, HNCO, HNCA, HNcoCA, or CBCAcoNH) for the target protein. On the other hand, we see the main practical applications of automated CST rather for source-target pairs with highly similar sequences such as orthologous proteins from different species and mutants in combination with temperature, pH, salt or other environment changes, ligand binding, <italic>etc.</italic>
</p>
<p>Future improvements of ARTINA-CST are possible, provided that more NMR data relevant for chemical shift transfer is collected and deposited in public repositories. It would allow for the use of statistical methods or machine learning to characterize chemical shift perturbation patterns resulting from different types of transfers (e.g., changes of the physical conditions or ligand binding).</p>
<p>We believe that the method can find future applications in fundamental studies of proteins with NMR spectroscopy, including investigations of protein structure, dynamics, and interactions. The method can be adopted easily by the NMR community and integrated in research protocols, as CST takes only up to 2&#xa0;h of computational time and the whole process can be carried out in the web browser using our cloud computing platform NMRtist (<ext-link ext-link-type="uri" xlink:href="https://nmrtist.org">https://nmrtist.org</ext-link>). Although this is a technical paper presenting a solution to a common problem in biomolecular NMR spectroscopy, where approaches analogous to molecular replacement in X-ray crystallography are not in common use, it may have a more general impact on biochemical research by simplifying the use of NMR in situations where it was so far considered a (too) laborious method. Additionally, we believe that the results presented here may serve as guidance for NMR practitioners who use the NMRtist platform (<xref ref-type="bibr" rid="B10">Klukowski et al., 2023</xref>), helping them to assess the practical benefits of the recently proposed ARTINA method in a new context.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: ETH Research Collection: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3929/ethz-b-000568621">https://doi.org/10.3929/ethz-b-000568621</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>HW performed research, prepared illustrations and tables, and wrote the draft manuscript. PK, RR, PG designed and supervised research. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>The study was supported by the EUREKA Eurostars grant E! 115328 and the Grant-in-Aid for Scientific Research 23K05660 of the Japan Society for the Promotion of Science. Open access funding was provided by ETH Zurich.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmolb.2023.1244029/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmolb.2023.1244029/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table2.pdf" id="SM2" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image2.pdf" id="SM3" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table3.PDF" id="SM4" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image1.pdf" id="SM5" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aeschbacher</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Blatter</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Maris</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Duss</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Allain</surname>
<given-names>F. H.-T.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Automated and assisted RNA resonance assignment using NMR chemical shift statistics</article-title>. <source>Nucleic Acids Res.</source> <volume>41</volume>, <fpage>e172</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt665</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aramini</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Rossi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y. J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Maglaqui</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Solution NMR structure of the NlpC/P60 domain of lipoprotein Spr from <italic>Escherichia coli</italic>: structural evidence for a novel cysteine peptidase catalytic triad</article-title>. <source>Biochemistry</source> <volume>47</volume>, <fpage>9715</fpage>&#x2013;<lpage>9717</lpage>. <pub-id pub-id-type="doi">10.1021/bi8010779</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Banelli</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Vuano</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fogolari</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Fusiello</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Esposito</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Corazza</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Automation of peak-tracking analysis of stepwise perturbed NMR spectra</article-title>. <source>J. Biomol. NMR</source> <volume>67</volume>, <fpage>121</fpage>&#x2013;<lpage>134</lpage>. <pub-id pub-id-type="doi">10.1007/s10858-017-0088-7</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bartels</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Billeter</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>G&#xfc;ntert</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>W&#xfc;thrich</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Automated sequence-specific NMR assignment of homologous proteins using the program GARANT</article-title>. <source>J. Biomol. NMR</source> <volume>7</volume>, <fpage>207</fpage>&#x2013;<lpage>213</lpage>. <pub-id pub-id-type="doi">10.1007/BF00202037</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bartels</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>G&#xfc;ntert</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Billeter</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>W&#xfc;thrich</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Garant - a general algorithm for resonance assignment of multidimensional nuclear magnetic resonance spectra</article-title>. <source>J. Comput. Chem.</source> <volume>18</volume>, <fpage>139</fpage>&#x2013;<lpage>149</lpage>. <pub-id pub-id-type="doi">10.1002/(sici)1096-987x(19970115)18:1&#x3c;139::aid-jcc13&#x3e;3.0.co;2-h</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>G&#xfc;ntert</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Buchner</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Combined automated NOE assignment and structure calculation with CYANA</article-title>. <source>J. Biomol. NMR</source> <volume>62</volume>, <fpage>453</fpage>&#x2013;<lpage>471</lpage>. <pub-id pub-id-type="doi">10.1007/s10858-015-9924-9</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>G&#xfc;ntert</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Mumenthaler</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>W&#xfc;thrich</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Torsion angle dynamics for NMR structure calculation with the new program DYANA</article-title>. <source>J. Mol. Biol.</source> <volume>273</volume>, <fpage>283</fpage>&#x2013;<lpage>298</lpage>. <pub-id pub-id-type="doi">10.1006/jmbi.1997.1284</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hoch</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Baskaran</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Burr</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Eghbalnia</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>Fujiwara</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Biological magnetic resonance Data Bank</article-title>. <source>Nucleic Acids Res.</source> <volume>51</volume>, <fpage>D368</fpage>&#x2013;<lpage>D376</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkac1050</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Combining automated peak tracking in SAR by NMR with structure-based backbone assignment from <sup>15</sup>N-NOESY</article-title>. <source>BMC Bioinforma.</source> <volume>13</volume>, <fpage>S4</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-13-S3-S4</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klukowski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Riek</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>G&#xfc;ntert</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>NMRtist: an online platform for automated biomolecular NMR spectra analysis</article-title>. <source>Bioinformatics</source> <volume>39</volume>, <fpage>btad066</fpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btad066</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klukowski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Riek</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>G&#xfc;ntert</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Rapid protein assignments and structures from raw NMR spectra with the deep learning technique ARTINA</article-title>. <source>Nat. Commun.</source> <volume>13</volume>, <fpage>6151</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-022-33879-5</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laveglia</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Giachetti</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cerofolini</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Haubrich</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Fragai</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ciulli</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Automated determination of nuclear magnetic resonance chemical shift perturbations in ligand screening experiments: the PICASSO web server</article-title>. <source>J. Chem. Inf. Model.</source> <volume>61</volume>, <fpage>5726</fpage>&#x2013;<lpage>5733</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.1c00871</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Orts</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gossert</surname>
<given-names>A. D.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Structure determination of protein-ligand complexes by NMR in solution</article-title>. <source>Methods</source> <volume>138</volume>, <fpage>3</fpage>&#x2013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymeth.2018.01.019</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Plata</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Utz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Werner</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Fully automated characterization of protein-peptide binding by microfluidic 2D NMR</article-title>. <source>J. Am. Chem. Soc.</source> <volume>145</volume>, <fpage>3204</fpage>&#x2013;<lpage>3210</lpage>. <pub-id pub-id-type="doi">10.1021/jacs.2c13052</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Redfield</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Robertson</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>1991</year>). &#x201c;<article-title>Assignment of the NMR spectra of homologous proteins</article-title>,&#x201d; in <source>Computational aspects of the study of biological macromolecules by nuclear magnetic resonance spectroscopy</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Hoch</surname>
<given-names>J. C.</given-names>
</name>
</person-group> (<publisher-loc>New York</publisher-loc>: <publisher-name>Plenum Press</publisher-name>), <fpage>303</fpage>&#x2013;<lpage>316</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schmidt</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>G&#xfc;ntert</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>A new algorithm for reliable and general NMR resonance assignment</article-title>. <source>J. Am. Chem. Soc.</source> <volume>134</volume>, <fpage>12817</fpage>&#x2013;<lpage>12829</lpage>. <pub-id pub-id-type="doi">10.1021/ja305091n</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thompson</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Sgourakis</surname>
<given-names>N. G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>G. H.</given-names>
</name>
<name>
<surname>Rossi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Y. F.</given-names>
</name>
<name>
<surname>Mills</surname>
<given-names>J. L.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Accurate protein structure modeling using sparse NMR data and homologous structure information</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>109</volume>, <fpage>9875</fpage>&#x2013;<lpage>9880</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1202485109</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H. Y.</given-names>
</name>
<name>
<surname>Neal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wishart</surname>
<given-names>D. S.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>RefDB: A database of uniformly referenced protein chemical shifts</article-title>. <source>J. Biomol. NMR</source> <volume>25</volume>, <fpage>173</fpage>&#x2013;<lpage>195</lpage>. <pub-id pub-id-type="doi">10.1023/a:1022836027055</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zieba</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Klukowski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gonczarek</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nikolaev</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Walczak</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Gaussian process regression for automated signal tracking in step-wise perturbed Nuclear Magnetic Resonance spectra</article-title>. <source>Appl. Soft Comput.</source> <volume>68</volume>, <fpage>162</fpage>&#x2013;<lpage>171</lpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2018.03.046</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>