<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mol. Biosci.</journal-id>
<journal-title>Frontiers in Molecular Biosciences</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mol. Biosci.</abbrev-journal-title>
<issn pub-type="epub">2296-889X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1386963</article-id>
<article-id pub-id-type="doi">10.3389/fmolb.2024.1386963</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Molecular Biosciences</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>DiffraGAN: a conditional generative adversarial network for phasing single molecule diffraction data to atomic resolution</article-title>
<alt-title alt-title-type="left-running-head">Matinyan et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fmolb.2024.1386963">10.3389/fmolb.2024.1386963</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Matinyan</surname>
<given-names>S.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2658751/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/Conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Filipcik</surname>
<given-names>P.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2722224/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>van Genderen</surname>
<given-names>E.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Abrahams</surname>
<given-names>J. P.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/938917/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Biozentrum</institution>, <institution>Basel University</institution>, <addr-line>Basel</addr-line>, <country>Switzerland</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Paul Scherrer Institute</institution>, <addr-line>Villigen</addr-line>, <country>Switzerland</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2088248/overview">Ruben Sanchez Garcia</ext-link>, University of Oxford, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2232952/overview">Jie E. Yang</ext-link>, University of Wisconsin-Madison, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1800002/overview">Daipayan Sarkar</ext-link>, Michigan State University, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: J. P. Abrahams, <email>jp.abrahams@unibas.ch</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1386963</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Matinyan, Filipcik, van Genderen and Abrahams.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Matinyan, Filipcik, van Genderen and Abrahams</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Proteins that adopt multiple conformations pose significant challenges in structural biology research and pharmaceutical development, as structure determination via single particle cryo-electron microscopy (cryo-EM) is often impeded by data heterogeneity. In this context, the enhanced signal-to-noise ratio of single molecule cryo-electron diffraction (simED) offers a promising alternative. However, a significant challenge in diffraction methods is the loss of phase information, which is crucial for accurate structure determination.</p>
</sec>
<sec>
<title>Methods</title>
<p>Here, we present DiffraGAN, a conditional generative adversarial network (cGAN) that estimates the missing phases at high resolution from a combination of single particle high-resolution diffraction data and low-resolution image data.</p>
</sec>
<sec>
<title>Results</title>
<p>For simulated datasets, DiffraGAN allows effectively determining protein structures at atomic resolution from diffraction patterns and noisy low-resolution images.</p>
</sec>
<sec>
<title>Discussion</title>
<p>Our findings suggest that combining single particle cryo-electron diffraction with advanced generative modeling, as in DiffraGAN, could revolutionize the way protein structures are determined, offering an alternative and complementary approach to existing methods.</p>
</sec>
</abstract>
<kwd-group>
<kwd>diffraction</kwd>
<kwd>cryo-EM</kwd>
<kwd>deep learning</kwd>
<kwd>generative adversarial network</kwd>
<kwd>simED</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Structural Biology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Single particle cryo-electron microcopy (cryo-EM) allows resolving the structure of macromolecular complexes with near atomic resolution (<xref ref-type="bibr" rid="B22">Zhang et al., 2008</xref>; <xref ref-type="bibr" rid="B19">Nakane et al., 2020</xref>; <xref ref-type="bibr" rid="B21">Yip et al., 2020</xref>). However, visualizing such biological complexes faces certain limitations. These include the very poor contrast of proteins, the need for low electron dose conditions to prevent radiation damage to the proteins, and the thickness of the ice encasing the specimens (<xref ref-type="bibr" rid="B3">Bepler et al., 2020</xref>). The expected signal-to-noise ratio (SNR) of a cryo-EM micrograph is estimated to be only as high as 0.1 (<xref ref-type="bibr" rid="B2">Baxter et al., 2009</xref>). While the SNR of an image can be improved by increasing the incident dose, this would destroy the macromolecule long before a sufficient number of scattering events is detected for a high-resolution structural analysis (<xref ref-type="bibr" rid="B18">Miao et al., 1999</xref>). These issues severely complicate structural analysis of small proteins, or of dynamic protein complexes that are present in many different conformations. In these cases, the signal becomes increasingly difficult to distinguish from noise.</p>
<p>We are analyzing far-field electron scattering diffraction data generated by diffracting a 10&#x2013;45&#xa0;nm narrow, parallel electron beam on a protein sample. Assuming the beam is not much wider than the size of the protein of interest, this approach is likely to improve the SNR ratio compared with cryo-EM imaging (<xref ref-type="bibr" rid="B17">Matinyan et al., 2023</xref>). Reportedly, much higher SNRs are observed when collecting data in this mode (<xref ref-type="bibr" rid="B14">Latychevskaia and Abrahams, 2019</xref>). However, measuring the diffracted wave function directly as a diffraction pattern has downsides.</p>
<p>In single particle cryo-EM the diffracted wave function is focused back into the image plane, providing phase information through phase contrast. The contrast observed in such images gives insights into the variations in electron density within the sample, and consequently, its structure (<xref ref-type="bibr" rid="B4">Clabbers and Abrahams, 2018</xref>). In the case of diffraction data, the phase information becomes much harder to retrieve, especially when the samples are complex molecules such as proteins. With the methodology outlined above, there is no easy and precise way to obtain the phase information of an electron wave function recorded in the diffraction plane. Here, we explore a computational approach for phase retrieval using neural networks, an approach uniquely suited to analysis of complex, multi-dimensional data. Neural networks of diverse topologies have been employed with great success in many areas of image analysis. Mirroring the architecture of biological neural networks, these computational models consist of interconnected neurons with learnable weights. Through iterative optimization, these networks are trained to minimize a loss function, aligning the model&#x2019;s predictions with a target domain. Among the various neural network architectures, convolutional neural networks (CNN) are particularly tailored for image data handling. Unlike standard feed-forward neural networks, CNNs incorporate specialized layers, such as convolution and pooling, to process spatial hierarchies in the input data. This design enables the network to recognize spatial patterns in the image, known as receptive fields, by selectively weighting neurons based on the significance of different portions of the input (<xref ref-type="bibr" rid="B15">LeCun et al., 1989</xref>).</p>
<p>Here, we have used conditional generative adversarial neural networks (GANs), which typically involve a pair of CNNs, with the purpose of generating the phase information that is missing in single molecule diffraction data (simED). We assumed an experimental setup in which diffraction data are collected by orthogonally scanning a sample with a narrow beam, and subsequently recording a low-resolution overview image of the scanned patch. We also assumed it would be feasible to correlate the diffraction patterns to locations in the image and identify which diffraction patterns belong to protein. We have established such an experimental setup, which combines software that is distributed by the hardware manufacturers JEOL, Amsterdam Scientific Instruments (ASI), and CEOS GmbH, and we are continuing to improve our setup. However, our setup is evolving rapidly and has not yet reached a stable state. In its current state, its technical details are beyond the scope of this paper, and by the time of publication would be superseded by improved versions. We trained a conditional GAN with simulated diffraction data of 20 different proteins, each in 1078 orientations, with the corresponding high-resolution projections as desired outcome (<xref ref-type="fig" rid="F1">Figure 1</xref>). Additionally, the network was given simulated defocused, low-resolution, noisy images of the proteins corresponding to each diffraction pattern. The resulting conditional GAN was successfully capable of phasing high-resolution diffraction data using noisy, low-resolution images of test proteins that were not included in the training.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Summary of DiffraGAN training procedure. Models of proteins were rotated to 1078 different angles, with a new pdb file of the protein saved at each angle. A diffraction pattern, high-resolution projection image, and low-resolution image of the protein in each of the poses were generated using multislice calculations from abTEM. These data were then used for DiffraGAN training.</p>
</caption>
<graphic xlink:href="fmolb-11-1386963-g001.tif"/>
</fig>
</sec>
<sec id="s2" sec-type="materials|methods">
<title>2 Materials and methods</title>
<p>Conditional GANs consist of two CNNs known as the generator and the discriminator. In standard GANs (<xref ref-type="bibr" rid="B7">Goodfellow et al., 2014</xref>), the generator&#x2019;s role is to learn how to convert a random noise vector <italic>x</italic> into an output image <italic>y</italic>. Conditional GANs, however, enhance this process by requiring the generator to learn from both a random noise vector and a specific input image. This method allows the generator to understand and replicate the structured aspects of the input, effectively penalizing any inaccuracies in the combined features of the generated output.</p>
<p>The discriminator, which is another CNN, plays a crucial role in evaluating the generator&#x2019;s outputs. It is trained adversarially to distinguish between genuine images and the &#x201c;fakes&#x201d; produced by the generator. The goal of the generator is to create images so convincing that the discriminator cannot tell them apart from real ones. This dynamic competition improves the generator&#x2019;s ability to produce highly realistic images, enhancing the overall performance of the conditional GAN. The objective of conditional GAN is to minimize the loss function expressed as:<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>G</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x003D;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced close="]" open="[" separators="&#x7c;">
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x002B;</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
</mml:mrow>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mfenced close="]" open="" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mfenced close="]" open="" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>where <italic>x</italic> is the observed image, <italic>y</italic> is the target image, and <italic>z</italic> is the random noise vector. C tries to minimize the objective against adversarial D, which tries to maximize it.</p>
<sec id="s2-1">
<title>2.1 The generator</title>
<p>In our generator, all data from the input image end up going through the narrowest part of the network, forming a &#x201c;U-Net&#x201d; architecture. Sometimes it can be beneficial for the GANs performance to skip the narrowest part(s) of the generator altogether by allowing straight connections between the early and late layers (<xref ref-type="bibr" rid="B10">Isola et al., 2016</xref>). Specifically, we added a skip connection between &#x3b9; and n-&#x3b9; layers, where n is the total number of the layers, thus forming a &#x201c;U-Net&#x201d; with skip connections. The common justification for allowing such skips, is that they may preserve and propagate larger elements in the input image&#x2019;s structure.</p>
</sec>
<sec id="s2-2">
<title>2.2 The discriminator</title>
<p>The discriminator evaluates the generated images and marks them as &#x201c;real&#x201d; or generated based on binary cross-entropy (BCE) loss. Our discriminator takes account the low-resolution and generated images, and the high-resolution projections. Inspired by PatchGAN (<xref ref-type="bibr" rid="B10">Isola et al., 2016</xref>), where the discriminator penalizes the structure at a scale of predefined patches, our discriminator also makes n number of decisions per image. It divides each image given to it into a specific number of &#x201c;patches&#x201d; of pixels, labeling each patch as &#x201c;real&#x201d; or generated. The final output of the discriminator is the average of all responses. We also implemented L1 loss to preserve low-resolution details. This term scales the loss of the generator according to the difference between corresponding &#x201c;real&#x201d; and generated pixels. The diagrams of the underlying discriminator and generator models are shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Pairs of diffraction patterns and low-resolution images were given to the generator that was designed to have a U-Net shape. The generator then generates an image conditioned on the input data. Both the generated and non-generated images are given to the discriminator. The discriminator labels each of these images as &#x201c;real&#x201d; or generated using the activation map. The more accurate the discriminator is at labelling images, the higher the loss of the generator. The objective of DiffraGAN is to minimize the loss of the generator, i.e., to make the generated images indistinguishable from the high-resolution images.</p>
</caption>
<graphic xlink:href="fmolb-11-1386963-g002.tif"/>
</fig>
</sec>
<sec id="s2-3">
<title>2.3 Datasets and data preparation</title>
<p>Below is a list of the proteins our conditional GAN (DiffraGAN) was trained on (the PDB entry ID for each protein is written in brackets next to it). The pdb files were downloaded from the Protein Data Bank (PDB) and fixed with the python package PDBFixer<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref> (<xref ref-type="bibr" rid="B5">Eastman et al., 2017</xref>). Specifically, we replaced nonstandard residues by their standard equivalents, removed all remaining heterogens, added missing hydrogen atoms and deleted water molecules. In cases the proteins were composed of multiple chains, only chain A was used. Each of these proteins was rotated around the center of mass, resulting in 1078 different rotation angle combinations. For each rotation, a new .pdb file was saved. 20 proteins were used for training and 5 proteins for test purposes (<xref ref-type="table" rid="T1">Table 1</xref>). The test proteins had no sequence similarity (they could not be aligned using the BLOSUM62 substitution matrix to account for evolutionary amino acid substitutions, with minimal gap opening and extension penalties) to the proteins that were used for DiffraGAN training.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The protein.pdb files that were divided into training and test parts. The saved DiffraGAN generator weights have been used to generate images from diffraction patterns and low-resolution images of the test proteins.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">PDB entries used</th>
<th align="center">ID</th>
<th align="center">Total structure weight</th>
<th align="center">Resolution</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center" colspan="4">Train dataset</td>
</tr>
<tr>
<td align="left">Apo CopZ from <italic>Bacillus subtilis</italic>
</td>
<td align="center">1P8G</td>
<td align="center">7.8&#xa0;kDa</td>
<td align="center">NMR (Conformer 1)</td>
</tr>
<tr>
<td align="left">Cyanobacterial copper metallochaperone, ScAtx1</td>
<td align="center">1SB6</td>
<td align="center">6.69&#xa0;kDa</td>
<td align="center">NMR (Conformer 1)</td>
</tr>
<tr>
<td align="left">C-terminal Domain (537&#x2013;610) of Human Heat Shock Protein 70</td>
<td align="center">2LMG</td>
<td align="center">8.38&#xa0;kDa</td>
<td align="center">NMR (Conformer 1)</td>
</tr>
<tr>
<td align="left">Ubiquitin</td>
<td align="center">1UBQ</td>
<td align="center">8.58&#xa0;kDa</td>
<td align="center">1.80&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Monoclinic turkey egg lysozyme</td>
<td align="center">135L</td>
<td align="center">14.23&#xa0;kDa</td>
<td align="center">1.30&#xc5;</td>
</tr>
<tr>
<td align="left">T4 lysozyme</td>
<td align="center">137L</td>
<td align="center">37.38&#xa0;kDa</td>
<td align="center">1.85&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Profilin I from <italic>Arabidopsis thaliana</italic>
</td>
<td align="center">1A0K</td>
<td align="center">14.28&#xa0;kDa</td>
<td align="center">2.20&#xc5;</td>
</tr>
<tr>
<td align="left">Peptidylprolyl isomerase, cyclophilin-like domain from <italic>Brugia malayi</italic>
</td>
<td align="center">1A33</td>
<td align="center">19.5&#xa0;kDa</td>
<td align="center">2.15&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Ornithine carbamoyltransferase from pyrococcus furiosus</td>
<td align="center">1A1S</td>
<td align="center">35.1&#xa0;kDa</td>
<td align="center">2.70&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Endoglucanase cel5a from <italic>bacillus</italic> agaradherans</td>
<td align="center">1A3H</td>
<td align="center">33.65&#xa0;kDa</td>
<td align="center">1.57&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Tyrosine phosphatase 1b</td>
<td align="center">1A5Y</td>
<td align="center">38.47&#xa0;kDa</td>
<td align="center">2.15&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Human UBC9</td>
<td align="center">1A3S</td>
<td align="center">18.17&#xa0;kDa</td>
<td align="center">2.80&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Cyclophilin from <italic>Brugia malayi</italic>
</td>
<td align="center">1A58</td>
<td align="center">19.5 kDA</td>
<td align="center">1.95&#xc5;</td>
</tr>
<tr>
<td align="left">Gamma s crystallin c-terminal domain</td>
<td align="center">1A7H</td>
<td align="center">20.64&#xa0;kDa</td>
<td align="center">2.56&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Fusarium solani cutinase</td>
<td align="center">1AGY</td>
<td align="center">20.83&#xa0;kDa</td>
<td align="center">1.15&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Ribonuclease A</td>
<td align="center">1AFU</td>
<td align="center">27.42&#xa0;kDa</td>
<td align="center">2.00&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Glutaminase-asparaginase of <italic>acinetobacter</italic> glutaminasificans</td>
<td align="center">1AGX</td>
<td align="center">35.52&#xa0;kDa</td>
<td align="center">2.90&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Top domain of african horse sickness virus vp7</td>
<td align="center">1AHS</td>
<td align="center">40.33&#xa0;kDa</td>
<td align="center">2.30&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Type I fructose 1,6-bisphosphate aldolase</td>
<td align="center">1ALD</td>
<td align="center">39.34&#xa0;kDa</td>
<td align="center">2.00&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Human CD40 ligand</td>
<td align="center">1ALY</td>
<td align="center">15.81&#xa0;kDa</td>
<td align="center">2.00&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="center" colspan="4">Test dataset</td>
</tr>
<tr>
<td align="left">Glutamate dehydrogenase</td>
<td align="center">1AUP</td>
<td align="center">49.21&#xa0;kDa</td>
<td align="center">2.50&#xc5;</td>
</tr>
<tr>
<td align="left">Proteinase K</td>
<td align="center">6CL8</td>
<td align="center">28.93&#xa0;kDa</td>
<td align="center">2.00&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Thaumatin</td>
<td align="center">5K7Q</td>
<td align="center">22.23&#xa0;kDa</td>
<td align="center">2.5&#xa0;&#xc5;</td>
</tr>
<tr>
<td align="left">Yeast Sti1 DP1 domain</td>
<td align="center">2LLV</td>
<td align="center">7.94&#xa0;kDa</td>
<td align="center">NMR (Conformer 1)</td>
</tr>
<tr>
<td align="left">Yeast Sti1 DP2 domain</td>
<td align="center">2LLW</td>
<td align="center">7.93&#xa0;kDa</td>
<td align="center">NMR (Conformer 1)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-4">
<title>2.4 Multislice simulation</title>
<p>The electron wave function describes the probability of finding an electron at a particular point in space. An electron wave function passing through matter, such as a protein, is diffracted, and analyzing such diffraction patterns can reveal the protein&#x2019;s structure. The diffraction pattern, the low-resolution and high-resolution projection image pairs of the protein in each of the .pdb files were created by multislice simulation as implemented in the abTEM package (<xref ref-type="bibr" rid="B16">Madsen and Susi, 2021</xref>). In abTEM, a complex array on a grid represents the plane wave function of the electron beam. An electron beam interacts with a specimen through the Coulomb potential of its electrons and nuclei. To calculate the electrostatic potential of the sample the independent atom model was used, which neglects any bonding effects and treats the sample as an array of atomic potentials. The wave function is passed slice-by-slice forward along the optical axis of the potential object, yielding an exit wave.</p>
<p>The absolute square of the discrete Fourier transform of the exit wave, yields the intensity distribution in diffraction plane. The high-resolution projection image was calculated as a convolution of the exit wave with a modelled CTF function with a defocus of &#x2212;50&#xa0;&#xc5;. The low-resolution, defocused images were simulated using a defocus value of &#x2212;1,000&#xc5; and an envelope function with a cutoff at 5&#xa0;mrad. The diffraction patterns contain information of up to 20&#xa0;mrad. The simulation and further image processing resulted in real space sampling of 0.27&#xa0;&#xc5; per pixel for high-resolution projection images. The low-resolution images were further degraded by including Poisson noise, which was imposed by altering the irradiation dose per &#xc5;<sup>2</sup> until they were almost indistinguishable from a typical cryo-EM particle-image, however with higher SNR (&#x223c;0.8 for the resolution bin of 8 to 5&#xa0;&#xc5;) (<xref ref-type="bibr" rid="B8">Heymann, 2022</xref>).</p>
<p>The resulting dataset consists of 25 &#xd7; 1078 diffraction, low-resolution and high-resolution image triplets, each triplet corresponding to one of the 25 proteins rotated by specific angle. Five of the proteins were used for validation and excluded from the training (<xref ref-type="table" rid="T1">Table 1</xref>). So far the training has been done with protein molecules that were simulated in vacuum. Details of the training of our DiffraGAN are described as in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<p>To create the results shown below, DiffraGAN was trained using the Adam gradient descent algorithm (<xref ref-type="bibr" rid="B12">Kingma and Ba, 2015</xref>), with a learning rate of 0.00002 and with &#x3b2;<sub>1</sub> parameter of 0.5. The training was performed on the sciCORE high-performance computing (HPC) platform of the University of Basel. Our final model was a 16-pixel patch DiffraGAN and further adjustments of the patch size did not improve the results significantly.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>3 Results</title>
<p>Since there is no objective loss function to train GANs, the performance of DiffraGAN had to be evaluated using the quality of the generated synthetic images. By considering different aspects of the images, from overall visual appearance to detailed statistical distributions, the evaluation process should provide a comprehensive assessment of the effectiveness of the image generation process. Qualitatively, it is clear when the generator is not working as expected when images do not correspond to the ground truth, and somebody observing two sets of generated images can tell which set matches the target set better &#x201c;by eye.&#x201d;</p>
<p>DiffraGAN was trained for 200 epochs using a data set of 20 proteins until the model reached an equilibrium. To evaluate the differences between high-resolution projected and corresponding generated images of the first test protein (PDB ID: 1AUP), a comprehensive comparison was conducted (Details of other test proteins are available in <xref ref-type="sec" rid="s10">Supplementary Material</xref>). Both sets of images were resized to a uniform resolution of 256 &#xd7; 256 pixels, and then randomly selected pairs were analyzed. The high-resolution projected and generated images were first visually compared to provide an initial assessment of their similarity (<xref ref-type="fig" rid="F3">Figure 3</xref>). To further quantify the differences, a mask was computed using a threshold (&#x394;pixel &#x003D; 10) on the smoothened absolute differences between corresponding pixels in the generated and projection images. This multi-step process captures the regions where pixel differences are most pronounced, offering a nuanced view of the spatial structure of discrepancies (<xref ref-type="fig" rid="F4">Figure 4</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Validation of DiffraGAN using diffraction and image data that were not used in the training. Top row: high-resolution diffraction patterns. Second row: low-resolution, defocused images. Third row: images generated by the generator using these inputs. Bottom row: target images. The axes show pixel number.</p>
</caption>
<graphic xlink:href="fmolb-11-1386963-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Generated, high-resolution projection image pairs and the detected edges of their differences. The last column depicts regions with an SSIM value &#x003e; 0.7. The axes show pixel number.</p>
</caption>
<graphic xlink:href="fmolb-11-1386963-g004.tif"/>
</fig>
<p>We also calculated structural similarity index (SSIM) as more perceptually relevant than other measures (<xref ref-type="bibr" rid="B1">Bakurov et al., 2022</xref>). SSIM incorporates perceptual phenomena, and is calculated as:<disp-formula id="equ3">
<mml:math id="m3">
<mml:mrow>
<mml:mtext>SSIM</mml:mtext>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x003D;</mml:mo>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x002B;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x002B;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x002B;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>Where: <inline-formula id="inf1">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the average of <italic>x</italic>, <inline-formula id="inf2">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the average of <italic>y</italic>, <inline-formula id="inf3">
<mml:math id="m6">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the variance of <italic>x</italic>, <inline-formula id="inf4">
<mml:math id="m7">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the variance of <italic>y</italic>, <italic>&#x3c3;</italic>
<sub>
<italic>x,y</italic> </sub>is the covariance of <italic>x</italic> and <italic>y</italic>, <inline-formula id="inf5">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf6">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x003D;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced close=")" open="(" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, are constants to stabilize the division with a weak denominator; <italic>L</italic> is the dynamic range of the pixel-values. The results are summarized in <xref ref-type="fig" rid="F4">Figure 4</xref>, which shows randomly sampled high-resolution projected and generated images. The SSIM values around the projection are notably low due to the projected images having a slight defocus, which renders SSIM particularly sensitive to fluctuations caused by defocus-induced phase reversals.</p>
<p>To quantitatively evaluate the degree of similarity between our generated images and their projected counterparts, we employed Fourier Ring Correlation (FRC) analysis (<xref ref-type="bibr" rid="B13">Koho et al., 2019</xref>). FRC provides a frequency domain metric for correlation at various scales between pairs of 2D images. Each image was Fourier transformed and the FRC was computed by normalizing the cross-spectrum of the two Fourier-transformed images by the geometric mean of their power spectra. A binary map was created to visualize the regions where the FRC values surpassed a 0.5 threshold (<xref ref-type="fig" rid="F5">Figure 5</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Fourier Ring Correlation (FRC) analysis of the high-resolution projection and generated images. For the purpose of this analysis, the binary mask threshold value for significant correlation was set at 0.5, allowing us to discern areas of high similarity between the compared images.</p>
</caption>
<graphic xlink:href="fmolb-11-1386963-g005.tif"/>
</fig>
<p>In addition to the 2D analysis, we computed the 1D FRC curves by averaging the FRC values over concentric rings in the frequency domain. We also used FRC analysis to reveal the information gain provided by the DiffraGAN by comparing the high-resolution projections (ground truth) with noisy and generated images (<xref ref-type="fig" rid="F6">Figure 6</xref>).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Fourier Ring Correlation (FRC) analysis of the high-resolution projections with noisy and generated images. First column: high-resolution diffraction patterns. Second column: low-resolution, defocused images. Third column: images generated by the generator using these inputs. Fourth column: FRC between ground truth (high-resolution projection) and noisy images. Fifth column: FRC between ground truth (high-resolution projection) and generated images.</p>
</caption>
<graphic xlink:href="fmolb-11-1386963-g006.tif"/>
</fig>
<p>The FRC analysis revealed a high degree of correlation at lower and medium spatial frequencies, as evidenced by the central region where the FRC exceeded this threshold. This observation suggests that the generated and projected images share significant structural features to a high resolution of approximately 1&#xc5;. The correlation diminished slowly at higher spatial frequencies which could be also caused by defocus induced phase reversals. The 1D FRC profiles confirmed the trends observed in the 2D analysis, with a drop in correlation coefficients beyond a certain spatial frequency, while still being high at 1&#xc5;, thereby quantitatively delineating the resolution limits of our generated images relative to their projected counterparts.</p>
<p>The average FRC, calculated using all 1078 angular orientations, drops below 0.5 at 0.90&#xa0;&#xc5; (<xref ref-type="fig" rid="F7">Figure 7</xref>). The combined visualizations and statistical analyses presented in <xref ref-type="fig" rid="F4">Figures 4</xref>&#x2013;<xref ref-type="fig" rid="F6">6</xref> confirm the quality and similarity of generated images in comparison to high-resolution projections.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Average FRC calculated from DiffraGAN-generated model images and ground truth high-resolution projection images of the first test protein (PDB ID: 1AUP). The FRC between model and ground truth does not reach zero at the highest resolution depicted in the graph, indicating that DiffraGAN managed to extract some phase information close to the Nyquist frequency.</p>
</caption>
<graphic xlink:href="fmolb-11-1386963-g007.tif"/>
</fig>
<p>8,000 high-resolution projection and DiffraGAN-generated images were then used for two 3D reconstructions of the simulated protein with RELION-4.0 (<xref ref-type="bibr" rid="B11">Kimanius et al., 2021</xref>). We reconstructed each set of high-resolution images <italic>ab initio</italic>. Both sets of images gave rise to a directly interpretable initial model and were refined to convergence using gold-standard refinement procedure without modification. As the generated images are not subject to CTF-related aberrations, we turned CTF correction off in <italic>ab initio</italic> model generation and in refinement. We fitted the PDB model in the maps using ChimeraX (<xref ref-type="bibr" rid="B6">Goddard et al., 2018</xref>) as shown in <xref ref-type="fig" rid="F8">Figure 8</xref>. DiffraGAN sometimes struggles to generate clear outer shape features and side chain distributions, which can translate into poor map/model fit on the periphery of the map, highlighted in square red dots. DiffraGAN occasionally introduces &#x201c;hot&#x201d; pixels in regions beyond the protein&#x2019;s structure, thereby possibly complicating the 3D refinement process. Nevertheless, the workflow creates a highly interpretable map that can be used for model fitting.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Maps created from generated and high-resolution projection images of the first test protein (PDB ID: 1AUP). Red squares represent absence of map density in case of generated images if present in case of projected ones.</p>
</caption>
<graphic xlink:href="fmolb-11-1386963-g008.tif"/>
</fig>
</sec>
<sec id="s4" sec-type="discussion">
<title>4 Discussion</title>
<p>GANs are a powerful class of deep learning models that have been applied to a wide range of tasks, including image generation and natural language processing. Recently, GANs have also been used for protein structure prediction (<xref ref-type="bibr" rid="B20">Strokach and Kim, 2022</xref>; <xref ref-type="bibr" rid="B9">Ingraham et al., 2023</xref>).</p>
<p>In this paper, we explore a new approach that could make phasing of coherent diffraction patterns of non-crystalline specimens much simpler, and possibly the same is true for more complex samples.</p>
<p>One key advantage of using conditional GANs for protein structure prediction from diffraction data and low-resolution features is their ability to capture the underlying feature distribution, which allows the model to generate diverse and realistic protein structures rather than just predicting a single most likely structure. This can be particularly useful for predicting the structures of proteins with multiple possible conformations, such as those involved in protein-protein interactions.</p>
<p>It appears that DiffraGAN can generate accurate images from corresponding single molecule diffraction patterns and low-resolution features of previously unseen proteins. All that is required, is a single molecule&#x2019;s diffraction pattern, its low-resolution image, and statistics concerning the general distribution of projected protein density. In our case, once DiffraGAN converged, it could map diffraction patterns and defocused images to the corresponding high-resolution image with reasonable accuracy. The model was reliably producing similar images for the same, slightly augmented input and corresponding to the high-resolution projection image, showcasing its stability and robustness in generating data (<xref ref-type="sec" rid="s10">Supplementary Figure S4</xref>). While there is theoretically no limit to the number of proteins the underlying GAN structure can be trained on, the performance of the GAN may decrease as the number of proteins that it is trained on increases. However, this will increase the generalization capacity of the network, as increasing the heterogeneity of the training data decreases the chances for overfitting. While theoretically such a function could be created, the increase in computational requirements could outweigh the potential benefits. It may be more beneficial to train multiple specialised GANs. This is equivalent to giving the GAN some additional information, like the molecular weight, or whether it concerns a membrane protein. How advantageous this approach could be has yet to be explored, however, from the test results (<xref ref-type="sec" rid="s10">Supplementary Figures S1, S2</xref>), it is clear that the trained GAN better maps the diffraction patterns from the proteins whose shape and size resembles that of most of the proteins it was trained on, while performing worse when trying to map the diffraction patterns from the protein whose shape is the furthest away from the rest.</p>
<p>DiffraGAN has certain limitations. Firstly, the protein models are simulated in a vacuum and the calculated diffraction data did not simulate Poisson noise due to counting statistics, so appropriate denoising should be ensured before employing this method. We chose defocus to be relatively low, because we had to use relatively small proteins to be able to do the simulation and training in a reasonable time frame. DiffraGAN been exclusively trained on simulated asymmetric proteins up to 40&#xa0;kDa in size and with a single defocus value. When tested with different defocus values, DiffraGAN can still reliably generate images from 1,500&#xa0;&#xc5; defocused images (<xref ref-type="sec" rid="s10">Supplementary Figures S5, S6</xref>); however, the performance drops significantly when the defocus is increased to 2,000&#xa0;&#xc5; (<xref ref-type="sec" rid="s10">Supplementary Figure S7</xref>). We anticipate that for larger protein complexes, higher defocus levels will produce similar results and providing the model with a training dataset that includes different defocus values could increase the model&#x2019;s reliability range. We are currently conducting tests to verify this assumption. We also anticipate that our current methodology, which has been optimized for smaller proteins, might require substantial modification to accommodate the unique requirements of membrane proteins. In addition, when DiffraGAN is applied to smaller proteins below 10&#xa0;kDa, as detailed in the <xref ref-type="sec" rid="s10">Supplementary Figure S3</xref>, and generated using the exact same procedure, there are more discrepancies between DiffraGAN results and the actual projections. This is because the same level of noise tends to obscure more details in smaller proteins compared to larger ones and uniform data generation and rescaling process disproportionately impacts the diffraction patterns of smaller proteins, often leading to a greater loss of resolution and detail. By using diffraction data collection in conjunction with cGAN image generation, structural analysis of proteins with electron microscopy can be extended to proteins that were previously infeasible to study using cryo-EM due to lack of SNR owing to their small size.</p>
<p>Our results suggest that phase extension from low resolution images to high resolution is feasible with electron diffraction data. Unlike traditional cryo-EM, which focuses on high-resolution image details, our goal is to capture those details in the diffraction patterns. We are developing a general package for single molecule electron diffraction (simED), in which first a sample is scanned with a narrow beam, and high-resolution diffraction data are collected on the fly. Then an overview of the sample is collected as an image, using a high dose and defocus, aiming for localization and low-resolution contours of any particles and phasing using DiffraGAN (<xref ref-type="sec" rid="s10">Supplementary Material S2</xref>). The additional information that allows this extension will in part be provided by the same restraints and constraints that allow phase extension in protein crystallography, like histogram matching, solvent flatness combined with the molecular contours and atomicity at high resolution. We speculate that DiffraGAN has discovered additional restraints in the data that are associated with macromolecular structures and may include elements of secondary structure and other complicated statistical correlations that are present in the data.</p>
<p>We are currently testing our methods on experimental data and are investigating to what extent it is limited by additional sources of experimental noise. Our <italic>in silico</italic> results are promising, indicating that the use of GANs for phasing may have the power to revolutionise methods in protein detection and structure elucidation by offering a potential solution to the phase problem.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The data and code presented in the study are deposited in the GitHub repository, accession link: <ext-link ext-link-type="uri" xlink:href="https://github.com/senikm/diffraGAN">https://github.com/senikm/diffraGAN</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>SM: Writing&#x2013;original draft, Writing&#x2013;review and editing, Conceptualization and Methodology. PF: Writing&#x2013;original draft, Writing&#x2013;review and editing. EvG: Writing&#x2013;original draft, Writing&#x2013;review and editing. JA: Conceptualization, Funding acquisition, Resources, Supervision, Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The authors declare that financial support was received for the research, authorship, and/or publication of this article. The following funding is acknowledged: HORIZON EUROPE Marie Sklodowska-Curie Actions (grant No. 956099 to SM); Schweizerischer Nationalfonds zur F&#xf6;rderung der Wissenschaftlichen Forschung (grant No. 205320_201012 to JA; grant No. TMPFP3_210216 to PF). Open access funding by PSI&#x2014;Paul Scherrer Institute.</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmolb.2024.1386963/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmolb.2024.1386963/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet2.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM2" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>
<ext-link ext-link-type="uri" xlink:href="https://github.com/openmm/pdbfixer">https://github.com/openmm/pdbfixer</ext-link>
</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bakurov</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Buzzelli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schettini</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Castelli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vanneschi</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Structural similarity index (SSIM) revisited: a data-driven approach</article-title>. <source>Expert Syst. Appl.</source> <volume>189</volume>, <fpage>116087</fpage>. <pub-id pub-id-type="doi">10.1016/J.ESWA.2021.116087</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baxter</surname>
<given-names>W. T.</given-names>
</name>
<name>
<surname>Grassucci</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Frank</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Determination of signal-to-noise ratios and spectral SNRs in cryo-EM low-dose imaging of molecules</article-title>. <source>J. Struct. Biol.</source> <volume>166</volume>, <fpage>126</fpage>&#x2013;<lpage>132</lpage>. <pub-id pub-id-type="doi">10.1016/J.JSB.2009.02.012</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bepler</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kelley</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Noble</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Berger</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Topaz-Denoise: general deep denoising models for cryoEM and cryoET</article-title>. <source>Nat. Commun. 2020</source> <volume>111</volume>, <fpage>5208</fpage>&#x2013;<lpage>5212</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-020-18952-1</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clabbers</surname>
<given-names>M. T. B.</given-names>
</name>
<name>
<surname>Abrahams</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Electron diffraction and three-dimensional crystallography for structural biology</article-title>. <source>Crystallogr. Rev.</source> <volume>24</volume>, <fpage>176</fpage>&#x2013;<lpage>204</lpage>. <pub-id pub-id-type="doi">10.1080/0889311X.2018.1446427</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eastman</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Swails</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chodera</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>McGibbon</surname>
<given-names>R. T.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Beauchamp</surname>
<given-names>K. A.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>OpenMM 7: rapid development of high performance algorithms for molecular dynamics</article-title>. <source>PLoS Comput. Biol.</source> <volume>13</volume>, <fpage>e1005659</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005659</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goddard</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>E. C.</given-names>
</name>
<name>
<surname>Pettersen</surname>
<given-names>E. F.</given-names>
</name>
<name>
<surname>Couch</surname>
<given-names>G. S.</given-names>
</name>
<name>
<surname>Morris</surname>
<given-names>J. H.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>UCSF ChimeraX: meeting modern challenges in visualization and analysis</article-title>. <source>Protein Sci.</source> <volume>27</volume>, <fpage>14</fpage>&#x2013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1002/PRO.3235</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goodfellow</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Pouget-Abadie</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mirza</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Warde-Farley</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ozair</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Generative adversarial networks</article-title>. <source>Commun. ACM</source> <volume>63</volume>, <fpage>139</fpage>&#x2013;<lpage>144</lpage>. <pub-id pub-id-type="doi">10.1145/3422622</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heymann</surname>
<given-names>J. B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The progressive spectral signal-to-noise ratio of cryo-electron micrograph movies as a tool to assess quality and radiation damage</article-title>. <source>Comput. Methods Programs Biomed.</source> <volume>220</volume>, <fpage>106799</fpage>. <pub-id pub-id-type="doi">10.1016/J.CMPB.2022.106799</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ingraham</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Baranov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Costello</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Barber</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ismail</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Illuminating protein space with a programmable generative model</article-title>. <source>Nat</source> <volume>623</volume>, <fpage>1070</fpage>&#x2013;<lpage>1078</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-023-06728-8</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Isola</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J. Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Efros</surname>
<given-names>A. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Image-to-Image translation with conditional adversarial networks</article-title>. <source>Proc. - 30th IEEE Conf. Comput. Vis. Pattern Recognit. CVPR 2017 2017-January</source>, <fpage>5967</fpage>&#x2013;<lpage>5976</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2017.632</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kimanius</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sharov</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Nakane</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Scheres</surname>
<given-names>S. H. W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>New tools for automated cryo-EM single-particle analysis in RELION-4.0</article-title>. <source>Biochem. J.</source> <volume>478</volume>, <fpage>4169</fpage>&#x2013;<lpage>4185</lpage>. <pub-id pub-id-type="doi">10.1042/BCJ20210708</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Adam: a method for stochastic optimization</article-title>,&#x201d; in <source>
<italic>3rd international Conference on learning representations, ICLR 2015 - conference track proceedings</italic>, (international conference on learning representations, ICLR)</source>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1412.6980v9">https://arxiv.org/abs/1412.6980v9</ext-link> (Accessed February 13, 2024)</comment>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koho</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tortarolo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Castello</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Deguchi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Diaspro</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vicidomini</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Fourier ring correlation simplifies image restoration in fluorescence microscopy</article-title>. <source>Nat. Commun.</source> <volume>10</volume>, <fpage>3103</fpage>. <pub-id pub-id-type="doi">10.1038/S41467-019-11024-Z</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Latychevskaia</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Abrahams</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Inelastic scattering and solvent scattering reduce dynamical diffraction in biological crystals</article-title>. <source>Acta Crystallogr. Sect. B Struct. Sci. Cryst. Eng. Mater.</source> <volume>75</volume>, <fpage>523</fpage>&#x2013;<lpage>531</lpage>. <pub-id pub-id-type="doi">10.1107/S2052520619009661</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Boser</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Denker</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Henderson</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Howard</surname>
<given-names>R. E.</given-names>
</name>
<name>
<surname>Hubbard</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>1989</year>). <article-title>Backpropagation applied to handwritten zip code recognition</article-title>. <source>Neural comput.</source> <volume>1</volume>, <fpage>541</fpage>&#x2013;<lpage>551</lpage>. <pub-id pub-id-type="doi">10.1162/NECO.1989.1.4.541</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Madsen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Susi</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The abTEM code: transmission electron microscopy from first principles</article-title>. <source>Open Res. Eur.</source> <volume>1</volume>, <fpage>24</fpage>. <pub-id pub-id-type="doi">10.12688/OPENRESEUROPE.13015.2</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matinyan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Demir</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Filipcik</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Abrahams</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Van Genderen</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Machine learning for classifying narrow-beam electron diffraction data</article-title>. <source>Acta Crystallogr. Sect. A Found. Adv.</source> <volume>79</volume>, <fpage>360</fpage>&#x2013;<lpage>368</lpage>. <pub-id pub-id-type="doi">10.1107/S2053273323004680</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Charalambous</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kirz</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sayre</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Extending the methodology of X-ray crystallography to allow imaging of micrometre-sized non-crystalline specimens</article-title>. <source>Nat</source> <volume>400</volume>, <fpage>342</fpage>&#x2013;<lpage>344</lpage>. <pub-id pub-id-type="doi">10.1038/22498</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nakane</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kotecha</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sente</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>McMullan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Masiulis</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>P. M. G. E.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Single-particle cryo-EM at atomic resolution</article-title>. <source>Nat</source> <volume>587</volume>, <fpage>152</fpage>&#x2013;<lpage>156</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2829-0</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Strokach</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>P. M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Deep generative modeling for protein design</article-title>. <source>Curr. Opin. Struct. Biol.</source> <volume>72</volume>, <fpage>226</fpage>&#x2013;<lpage>236</lpage>. <pub-id pub-id-type="doi">10.1016/J.SBI.2021.11.008</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yip</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Paknia</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Chari</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Stark</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Atomic-resolution protein structure determination by cryo-EM</article-title>. <source>Nat</source> <volume>587</volume>, <fpage>157</fpage>&#x2013;<lpage>161</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-020-2833-4</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Settembre</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dormitzer</surname>
<given-names>P. R.</given-names>
</name>
<name>
<surname>Bellamy</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Harrison</surname>
<given-names>S. C.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Near-atomic resolution using electron cryomicroscopy and single-particle reconstruction</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>105</volume>, <fpage>1867</fpage>&#x2013;<lpage>1872</lpage>. <pub-id pub-id-type="doi">10.1073/PNAS.0711623105</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>