<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2024.1407470</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Interpretable deep learning reveals the role of an E-box motif in suppressing somatic hypermutation of AGCT motifs within human immunoglobulin variable regions</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Tambe</surname>
<given-names>Abhik</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2692144"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>MacCarthy</surname>
<given-names>Thomas</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/405663"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Pavri</surname>
<given-names>Rushad</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2695770"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Biochemistry and Cell Biology, Stony Brook University</institution>, <addr-line>Stony Brook, NY</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Applied Mathematics and Statistics, Stony Brook University</institution>, <addr-line>Stony Brook, NY</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Research Institute of Molecular Pathology (IMP)</institution>, <addr-line>Vienna</addr-line>, <country>Austria</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Peter Gorer Department of Immunobiology, School of Immunology &amp; Microbial Sciences, King&#x2019;s College London</institution>, <addr-line>London</addr-line>, <country>United Kingdom</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Amy L. Kenter, University of Illinois Chicago, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Robert W. Maul, National Institute on Aging (NIH), United States</p>
<p>Alberto Martin, University of Toronto, Canada</p>
<p>Paolo Casali, The University of Texas Health Science Center at San Antonio, United States</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Rushad Pavri, <email xlink:href="mailto:rushad.pavri@kcl.ac.uk">rushad.pavri@kcl.ac.uk</email>
</p>
</fn>
<fn fn-type="deceased" id="fn003">
<p>&#x2020;Deceased</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>28</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1407470</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Tambe, MacCarthy and Pavri</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Tambe, MacCarthy and Pavri</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Somatic hypermutation (SHM) of immunoglobulin variable (V) regions by activation induced deaminase (AID) is essential for robust, long-term humoral immunity against pathogen and vaccine antigens. AID mutates cytosines preferentially within WRCH motifs (where W=A or T, R=A or G and H=A, C or T). However, it has been consistently observed that the mutability of WRCH motifs varies substantially, with large variations in mutation frequency even between multiple occurrences of the same motif within a single V region. This has led to the notion that the immediate sequence context of WRCH motifs contributes to mutability. Recent studies have highlighted the potential role of local DNA sequence features in promoting mutagenesis of AGCT, a commonly mutated WRCH motif. Intriguingly, AGCT motifs closer to 5&#x2019; ends of V regions, within the framework 1 (FW1) sub-region1, mutate less frequently, suggesting an SHM-suppressing sequence context.</p>
</sec>
<sec>
<title>Methods</title>
<p>Here, we systematically examined the basis of AGCT positional biases in human SHM datasets with DeepSHM, a machine-learning model designed to predict SHM patterns. This was combined with integrated gradients, an interpretability method, to interrogate the basis of DeepSHM predictions.</p>
</sec> <sec>
<title>Results</title>
<p>DeepSHM predicted the observed positional differences in mutation frequencies at AGCT motifs with high accuracy. For the conserved, lowly mutating AGCT motifs in FW1, integrated gradients predicted a large negative contribution of 5&#x2019;C and 3&#x2019;G flanking residues, suggesting that a CAGCTG context in this location was suppressive for SHM. CAGCTG is the recognition motif for E-box transcription factors, including E2A, which has been implicated in SHM. Indeed, we found a strong, inverse relationship between E-box motif fidelity and mutation frequency. Moreover, E2A was found to associate with the V region locale in two human B cell lines. Finally, analysis of human SHM datasets revealed that naturally occurring mutations in the 3&#x2019;G flanking residues, which effectively ablate the E-box motif, were associated with a significantly increased rate of AGCT mutation.</p>
</sec> <sec>
<title>Discussion</title>
<p>Our results suggest an antagonistic relationship between mutation frequency and the binding of E-box factors like E2A at specific AGCT motif contexts and, therefore, highlight a new, suppressive mechanism regulating local SHM patterns in human V regions.</p>
</sec>
</abstract>
<kwd-group>
<kwd>somatic hypermutation (SHM)</kwd>
<kwd>activation induced deaminase (AID)</kwd>
<kwd>immunoglobulin heavy chain</kwd>
<kwd>deep learning</kwd>
<kwd>integrated gradients</kwd>
<kwd>E-box transcription factors</kwd>
<kwd>E2A</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="2"/>
<equation-count count="0"/>
<ref-count count="76"/>
<page-count count="12"/>
<word-count count="5779"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>B Cell Biology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>Somatic hypermutation (SHM) of immunoglobulin (IG) genes in B cells is essential for producing high-affinity antibodies against antigens on pathogens and vaccines (<xref ref-type="bibr" rid="B1">1</xref>). SHM occurs within germinal centers of secondary lymphoid tissue where iterative cycles of mutation and antigen-mediated affinity selection result in the clonal expansion of B cells expressing antibodies with higher affinity towards the target antigen (<xref ref-type="bibr" rid="B2">2</xref>). Point mutations are introduced into the variable (V) region of the IG heavy chain (<italic>IGH</italic>) and light chain genes by the enzyme, activation-induced deaminase (AID) (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B4">4</xref>), which deaminates cytosine to uracil on single-stranded DNA (ssDNA) in a transcription-dependent manner (<xref ref-type="bibr" rid="B5">5</xref>&#x2013;<xref ref-type="bibr" rid="B8">8</xref>). AID preferentially acts on WRCH hotspots (where W=A/T, R=A/G and H=A/C/T) (<xref ref-type="bibr" rid="B9">9</xref>&#x2013;<xref ref-type="bibr" rid="B11">11</xref>). The U:G mismatch can result in a C&gt;T transition mutation upon replication, while induction of error-prone repair mechanisms such as base excision repair can lead to C&gt;G or C&gt;A transversions (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B13">13</xref>). Additionally, mismatch repair pathways generate mutations at A/T residues surrounding the U:G mismatch (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B13">13</xref>). A striking and consistent feature of SHM profiles is the differential mutability of WRCH motifs wherein mutation frequencies of WRCH motifs vary substantially, not only between different motifs but also between multiple occurrences of identical motifs within a V region (<xref ref-type="bibr" rid="B14">14</xref>&#x2013;<xref ref-type="bibr" rid="B18">18</xref>). This has led to the idea that the sequence context of these motifs plays a major role in determining their mutability (<xref ref-type="bibr" rid="B14">14</xref>&#x2013;<xref ref-type="bibr" rid="B18">18</xref>). This idea has recently received important experimental support from a study which showed that the density of pyrimidine dimers (PyPy) in the 6 nt region upstream of AGCT motifs correlates with increased mutability, perhaps because PyPy richness confers flexibility to ssDNA that may facilitate AID targeting (<xref ref-type="bibr" rid="B19">19</xref>). Therefore, a major effort in the field is to further understand the mechanisms regulating differential mutability during SHM.</p>
<p>The recruitment of AID to V regions and other IG and non-IG targets has been linked to specific activating chromatin modifications (<xref ref-type="bibr" rid="B20">20</xref>&#x2013;<xref ref-type="bibr" rid="B29">29</xref>) and transcriptional and co-transcriptional activities, notably, RNA polymerase II pausing (<xref ref-type="bibr" rid="B30">30</xref>&#x2013;<xref ref-type="bibr" rid="B36">36</xref>), RNA exosome-mediated processing of RNA: DNA hybrids (<xref ref-type="bibr" rid="B37">37</xref>, <xref ref-type="bibr" rid="B38">38</xref>) and convergent transcription (<xref ref-type="bibr" rid="B39">39</xref>). However, nascent transcriptional profiling of multiple V regions and hundreds of non-IG AID target loci at single-nucleotide resolution revealed no apparent correlation between mutation frequency of specific WRCH motifs and transcriptional strength or transcriptional features in its neighborhood (<xref ref-type="bibr" rid="B40">40</xref>). Thus, although transcriptional activities and chromatin marks are important for recruiting AID to its genomic targets, the observed differential mutability characteristic of SHM patterns cannot be explained solely by the transcriptional landscape (<xref ref-type="bibr" rid="B40">40</xref>). This finding further supports the notion that, following AID recruitment to V regions, the relative mutation frequency of WRCH motifs likely depends on the sequence neighborhood of each motif.</p>
<p>The major <italic>cis</italic>-regulatory elements regulating SHM are the IG enhancers which harbor binding sites for a plethora of transcription factors (TFs) (<xref ref-type="bibr" rid="B41">41</xref>&#x2013;<xref ref-type="bibr" rid="B45">45</xref>). Amongst these, the E-box-binding TF, E2A, has been linked to SHM in multiple studies (<xref ref-type="bibr" rid="B42">42</xref>, <xref ref-type="bibr" rid="B46">46</xref>&#x2013;<xref ref-type="bibr" rid="B50">50</xref>). In experiments of enhancer-driven SHM of reporter substrates, elements with the E-box motif were found to have a particularly large impact on SHM, and among the TFs predicted to bind, loss of E2A was shown to cause a significant decrease in SHM (<xref ref-type="bibr" rid="B43">43</xref>). E2A, AID and other TFs were reported to form a complex that could associate with <italic>IGH</italic> (<xref ref-type="bibr" rid="B51">51</xref>, <xref ref-type="bibr" rid="B52">52</xref>). It has also been shown that the presence of a E2A-binding motif enhances SHM in nearby regions (<xref ref-type="bibr" rid="B53">53</xref>) and may even facilitate AID recruitment (<xref ref-type="bibr" rid="B54">54</xref>).</p>
<p>To understand the mechanisms of differential mutability, our group recently developed DeepSHM, a convolutional neural network model trained to predict mutation frequencies of the central nucleotide in a 5-mer, 9-mer, 15-mer or 21-mer motif derived from human SHM data (<xref ref-type="bibr" rid="B18">18</xref>). The model achieved a high cross-validated Pearson correlation of r=0.81 with the experimental data (<xref ref-type="bibr" rid="B23">23</xref>). Moreover, model performance did not improve beyond a 15-mer context, that is, a 21-mer context did not significantly improve the predictions (<xref ref-type="bibr" rid="B18">18</xref>). In addition to the advantage brought forth by the expanded sequence context, compared to previous work which used shorter window sizes of 5&#x2013;7 nucleotides (<xref ref-type="bibr" rid="B55">55</xref>), this approach also allows for use of interpretability techniques, which can be used to understand the model&#x2019;s reasoning behind its predictions and, therefore, gain insights into potential biological mechanisms (<xref ref-type="bibr" rid="B18">18</xref>).</p>
<p>In this study, we extend the use of interpretability methods on DeepSHM to investigate the basis for positional differences in mutability of AGCT, one of the most frequently mutated WRCH motifs in human V regions. We report that conserved AGCTs near the 5&#x2019; end of V regions undergo significantly lower SHM than other AGCTs and that this suppression of mutability coincides with the presence of an E2A-binding E-box motif. We find that E2A is associated with V regions. The negative impact of E-box motifs was independent of the positive effect of PyPy richness. Ablation of this motif through naturally occurring mutations correlated with significantly increased mutation frequency. Thus, our study highlights a potential mechanism by which local sequence context negatively regulates mutability and contributes to the discrete SHM profiles of V regions.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="s2_1">
<title>Sequence data</title>
<p>The 15-mer sequence dataset used to train DeepSHM was generated as described in our previous publication (<xref ref-type="bibr" rid="B18">18</xref>) and is available at <ext-link ext-link-type="uri" xlink:href="https://gitlab.com/maccarthyslab/deepshm">https://gitlab.com/maccarthyslab/deepshm</ext-link>. Germline IGHV reference sequences from the international ImMunoGeneTics information system (IMGT) (<xref ref-type="bibr" rid="B56">56</xref>) were downloaded and split into k-mers using a sliding window approach. Mutation frequencies for the central nucleotide in each k-mer were calculated by comparing against a B cell receptor (BCR) sequencing (BCR-seq) dataset from marginal zone, memory, and plasma B cells from healthy volunteers, as described in our previous study (<xref ref-type="bibr" rid="B57">57</xref>). To study intrinsic SHM patterns and avoid confounding issues arising from clonal selection in germinal centers, we used only non-productive sequences (containing internal frameshifts or stop codons) and clonally independent sequences (one sequence per clone, as assigned by Change-O (<xref ref-type="bibr" rid="B58">58</xref>), which uses CDR3s to segment clones) (<xref ref-type="bibr" rid="B57">57</xref>).</p>
<p>For this study, only 15-mers containing an AGCT motif, with either the G or the C as the central nucleotide, were used with DeepSHM to predict mutation frequencies. The total 15-mer dataset was processed to extract those containing AGCTs using custom Python scripts. Our statistical analysis of synonymous mutations (those that do not change the protein sequence) ablating the CAGCTG motif was done with productive, clonally independent sequences. All statistical tests were performed using SciPy (<xref ref-type="bibr" rid="B59">59</xref>).</p>
</sec>
<sec id="s2_2">
<title>DeepSHM</title>
<p>DeepSHM (<ext-link ext-link-type="uri" xlink:href="https://gitlab.com/maccarthyslab/deepshm">https://gitlab.com/maccarthyslab/deepshm</ext-link>) is a deep learning model that uses a convolutional neural network architecture to predict mutation frequency or substitution rate of the central nucleotide in a k-mer of size 5, 9, 15 or 21. We used the 15-mer mutation frequency model, which takes a DNA sequence of 15 nucleotides as input and outputs a predicted mutation frequency value between 0 and 1. The 15-mer sequences were encoded into a 4 x 15 binary matrix, with rows corresponding to the 4 nucleotides and columns to the 15 positions along the k-mer. In each column, a 1 was placed in the appropriate row to denote the base identity for that position, while the remaining rows were 0s. This procedure, called one-hot encoding, is a common method for converting categorical data (A, G, C, T) into a machine-readable format (0s and 1s).</p>
<p>We downloaded the h5 file containing the model (model_15_mf.h5) and used it with Python to predict mutation frequencies for AGCT 15-mers in our dataset.</p>
</sec>
<sec id="s2_3">
<title>Integrated gradients</title>
<p>Integrated gradients is an attribution method that measures the impact of individual inputs towards the output prediction of a deep learning model (<xref ref-type="bibr" rid="B60">60</xref>). It relies on a baseline value to compute a path integral of the model&#x2019;s gradients with respect to its inputs, from the baseline to the input value. Since our input data is binary, we used a zero-matrix as our baseline with 50 steps taken from baseline to input. All 15 nucleotides in the 15-mer are considered as input features in the prediction of the mutation frequency of the central nucleotide, hence integrated gradients calculates a score for each base according to its impact on the output prediction.</p>
<p>The following GitHub repository was used to compute the integrated gradients scores for each of the DeepSHM predictions: <ext-link ext-link-type="uri" xlink:href="https://github.com/hiranumn/IntegratedGradients">https://github.com/hiranumn/IntegratedGradients</ext-link>. The repository was cloned and imported into the python script where the DeepSHM predictions were being run and used to compute integrated gradients scores for each input in each prediction. We generated sequence logo plots to visualize the frequency of nucleotides occurring at each position across 15-mers in each subregion using Logomaker (<xref ref-type="bibr" rid="B61">61</xref>), which is available at the following GitHub repository: <ext-link ext-link-type="uri" xlink:href="https://github.com/jbkinney/logomaker">https://github.com/jbkinney/logomaker</ext-link>.</p>
</sec>
<sec id="s2_4">
<title>MOODS</title>
<p>MOODS (<ext-link ext-link-type="uri" xlink:href="https://github.com/jhkorhonen/MOODS">https://github.com/jhkorhonen/MOODS</ext-link>) is a position-weight matrix (PWM) matching algorithm that takes sequences and a counts matrix as inputs and outputs match scores for a segment of the sequences (<xref ref-type="bibr" rid="B62">62</xref>). The counts matrix is a 4&#xd7;n matrix where the rows correspond to nucleotides (A, G, C, or T) and the columns correspond to positions along the TF binding motif, with the number of counts for each nucleotide in each position empirically obtained using SELEX and available on the JASPAR database (<xref ref-type="bibr" rid="B63">63</xref>). MOODS uses log-likelihood scoring to convert the counts matrix to a PWM, which it then compares against the sequence to generate a match score, only reporting scores at positions that exceed a <italic>P</italic> value cutoff of 0.001. We used MOODS to gauge the fidelity of our 15-mer sequences to binding motifs for the E-box TFs, E2A (<ext-link ext-link-type="uri" xlink:href="https://jaspar2020.genereg.net/matrix/MA0522.2/">https://jaspar2020.genereg.net/matrix/MA0522.2/</ext-link>) and TFAP4 (<ext-link ext-link-type="uri" xlink:href="https://testjaspar.uio.no/matrix/MA1570.1/">https://testjaspar.uio.no/matrix/MA1570.1/</ext-link>).</p>
</sec>
<sec id="s2_5">
<title>ChIP-seq data</title>
<p>The E2A ChIP-seq data in Ramos cells was taken from a previous study (<xref ref-type="bibr" rid="B45">45</xref>) and is available at <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/587064">https://www.ncbi.nlm.nih.gov/bioproject/587064</ext-link>. The E2A ChIP-seq data in GM12878 cells was taken from ENCODE (<xref ref-type="bibr" rid="B64">64</xref>) and is available at <ext-link ext-link-type="uri" xlink:href="https://www.encodeproject.org/experiments/ENCSR000BQT/">https://www.encodeproject.org/experiments/ENCSR000BQT/</ext-link>. Both datasets were subject to the same analysis pipeline - a local alignment to hg38 of both case and control datasets using bowtie2 (<xref ref-type="bibr" rid="B65">65</xref>), sorting and indexing the resulting bam file using samtools (<xref ref-type="bibr" rid="B66">66</xref>). The callpeaks function of MACS2 (<xref ref-type="bibr" rid="B67">67</xref>) was run on the aligned bam files using the default q-value cutoff of 0.05 to call peaks. These peaks were then subject to motif enrichment analysis using the findMotifsGenome.pl function of the HOMER suite (<xref ref-type="bibr" rid="B68">68</xref>), using the hg38 genome and default size setting of 200 bp. The RPKM calculation was conducted using deeptools bamCoverage (<xref ref-type="bibr" rid="B69">69</xref>) with the bin size parameter set to 500 bp.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s3_1">
<title>DeepSHM recapitulates the observed positional differences in AGCT mutability</title>
<p>V regions can be structurally divided into antigen-binding complementarity-determining regions (CDR1&#x2013;3) and intervening structural framework regions (FW1&#x2013;3). AGCT is one of the most highly mutated WRCH motifs in V regions (<xref ref-type="bibr" rid="B70">70</xref>). The fact that AGCT is palindromic increases the probability that AID will deaminate cytosines on both the forward and the reverse strands (<xref ref-type="bibr" rid="B70">70</xref>). All human IGHV genes (at least the *01 IMGT alleles), except for three from the IGHV2 family, have one or more AGCT motifs near the 5&#x2019; end located in FW1 (<xref ref-type="bibr" rid="B57">57</xref>).</p>
<p>Our previous analysis of IGHV3&#x2013;23*01 non-productive sequences showed higher mutability of AGCTs in the CDRs and a particularly low mutability of the 5&#x2019; AGCT in FW1 (<xref ref-type="bibr" rid="B14">14</xref>). To examine this differential mutability of AGCT motifs across all IGHV genes, we used DeepSHM to predict mutation frequencies of the central nucleotides in the AGCT 15-mers in our dataset and compared the results with the observed data using a correlation analysis. We achieved a Pearson correlation of r=0.92 for those with a central G site (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>) and r=0.86 for those with a central C (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>). These high correlations suggested that the neural network had identified sequence features that distinguish low from high mutation frequencies for AGCT sites. To confirm that the positional differences in AGCT mutability previously observed for IGHV3&#x2013;23*01 applied to other IGHV alleles, we examined our dataset of 16,870 15-mers across 65 IGHV alleles and their associated mutation frequencies (<xref ref-type="bibr" rid="B57">57</xref>). We separated the AGCT 15-mers in our dataset by IMGT subregion and plotted the observed and predicted mutation frequencies for those with central Cs (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1C</bold>
</xref>) and central Gs (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1D</bold>
</xref>). We observed a statistically significant difference in mutation frequency between the AGCT motifs in FW1 and all other subregions for both observed and predicted datasets centered on the G (t-test, <italic>P</italic>&lt;10<sup>&#x2013;30</sup>) (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1C</bold>
</xref>) and C (t-test, <italic>P</italic>&lt;10<sup>&#x2013;20</sup>) (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1D</bold>
</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>DeepSHM model performance on AGCT 15-mers. <bold>(A, B)</bold> Correlation scatter plots between observed and DeepSHM-predicted mutation frequencies. 15-mers centered on G (AGCT) <bold>(A)</bold> or C (AGCT) <bold>(B)</bold>. Each dot represents a 15-mer, the black line is the x=y diagonal and the red line indicates the best fit with intercept and coefficient computed using a linear regression. The r value is the Pearson correlation coefficient, and the <italic>P</italic> value is computed using a Wald test. <bold>(C, D)</bold> Violin plots showing the distributions of observed (white) and DeepSHM-predicted (blue) mutation frequencies for AGCT 15-mers within CDR and FW regions centered on G (AGCT) <bold>(C)</bold> and C (AGCT) <bold>(D)</bold>. The white dots represent the median, the black boxes show the interquartile range, and the whiskers encapsulate points that fall between 1.5 times the inter-quartile range.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1407470-g001.tif"/>
</fig>
<p>We conclude that that AGCTs in FW1 are significantly less mutated than those in other V subregions and that DeepSHM can recapitulate these observed positional differences in AGCT mutability.</p>
</sec>
<sec id="s3_2">
<title>Integrated gradients reveals a sequence context associated with decreased mutability of FW1 AGCT motifs</title>
<p>To interrogate the specific sequence features associated with high or low mutation frequency predictions, we used an interpretability method, integrated gradients (<xref ref-type="bibr" rid="B22">22</xref>). Integrated gradients analyses involve computing the derivative of the output (mutation frequency prediction) with respect to the input (15-mer sequence) to ascribe importance to input features based on their impact on the output prediction. Specifically, a higher integrated gradients score would imply that a small change in input had a more positive contribution towards the output prediction. Conversely, lower integrated gradients scores indicate that changes in input features contributed negatively to the predicted output.</p>
<p>We generated integrated gradients scores for each nucleotide within the AGCT 15-mers for its prediction of (i.e. contribution towards) the mutation frequency of the central G or C within the hotspot. We plotted the range of scores for each position as boxplots which were further categorized based on the location of the 15-mers within FW and CDR subregions (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2A&#x2013;J</bold>
</xref>).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Integrated gradients scores for each nucleotide in AGCT 15-mers across V subregions shown as boxplots. The left column consists of sequences with a central G <bold>(A&#x2013;E)</bold> and the right column consists of sequences with a central C <bold>(F&#x2013;J)</bold>. Rows correspond to the indicated V subregion and the sequence logo below each boxplot corresponds to the nucleotide frequency at each position. The boxes represent the inter-quartile region of the distribution of integrated gradient scores for each nucleotide, with the black line through the box showing the median score and the whiskers representing 1.5 times the inter-quartile range. Outlier points are shows as dots. Note that nucleotides in the central AGCT hotspot (boxed) tend to have the largest scores in the 15-mer and that the 5&#x2019; and 3&#x2019; flanking nucleotides for the FW1 AGCT <bold>(F)</bold> have the lowest scores in the 15-mer.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1407470-g002.tif"/>
</fig>
<p>As a positive control, we observed that nucleotides within the AGCT hotspot across all central G 15-mers (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2A&#x2013;E</bold>
</xref>) and 85% of central C 15-mers (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2F&#x2013;J</bold>
</xref>) had a positive integrated gradients score, meaning that the presence of these nucleotides increased the mutability prediction of the central G or C. We then examined the integrated gradients scores for FW1 AGCTs (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2A, F</bold>
</xref>) as they are significantly less mutated than AGCTs in other V sub-regions (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1C</bold>
</xref>). We found that 78% of the lowly mutating FW1 AGCTs were flanked by a 5&#x2019;-C and 3&#x2019;-G nucleotide, both of which have large negative integrated gradient scores (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2F</bold>
</xref>). This suggests that an extended CAGCTG motif context decreases the mutability prediction for the central G and C nucleotides within these lowly mutating FW1 AGCTs, implying that CAGCTG motifs may be less frequently targeted by AID.</p>
<p>To explore this idea further, we directly compared integrated gradients scores of the 5&#x2019; and 3&#x2019; nucleotides flanking AGCTs across all 15-mers. We found that for CAGCTG-containing 15-mers, the integrated gradients scores were almost always negative for the 5&#x2019;-C (98%) and consistently negative for the 3&#x2019;-G (100%), supporting the notion that the CAGCTG context has a predominantly negative influence on mutagenesis of AGCT (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>). Additionally, of all AGCT flanking nucleotide combinations, the 5&#x2019;-C and 3&#x2019;-G combinations were overwhelmingly within the FW1 region and were significantly less mutated than AGCTs with other flanking nucleotide combinations (Mann-Whitney test, <italic>P</italic>&lt;10<sup>&#x2013;30</sup>) (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>). We note that although most of the FW1 AGCT motifs were flanked by 5&#x2019;-C and 3&#x2019;-G nucleotides, even those flanked by other nucleotide combinations tended to have lower mutation frequencies (blue dots in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>). This suggests that the position of the AGCT within the V region may also have some influence on its mutability. In addition, a fraction of CAGCTG motifs in FW2 undergo higher mutation frequency than those in FW1 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>), suggesting the presence of additional mechanisms, possibly involving differences in the larger sequence context of these motifs, that influence differential mutability, which we address in the following section.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Integrated gradients scores for the 5&#x2019; and 3&#x2019; flanking nucleotides of AGCT motifs across all human V regions shown as a scatter plot. Each dot/cross corresponds to a 15-mer. 15-mers in which the central AGCTs are flanked by 5&#x2019;-C and 3&#x2019;-G (CAGCTG motifs) are indicated with a cross (x).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1407470-g003.tif"/>
</fig>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Swarm plot depicting mutation frequencies for the central G and C residues within AGCT 15-mers categorized based on the identity of the 5&#x2019; and 3&#x2019; nucleotides flanking AGCT. The color coding highlights the location of the 15-mer in CDRs or FWs. Each AGCT is represented by two dots - one for the central C and one for the central G. AGCT motifs flanked by 5&#x2019;-C and 3&#x2019;-G, corresponding to the CAGCTG motif (first category on the left), has a significantly lower mutation frequency (<italic>P</italic>&lt;10<sup>&#x2013;30</sup>) than any other pair as computed by a Mann-Whitney U Test.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1407470-g004.tif"/>
</fig>
<p>We conclude that the weakly mutated AGCTs in FW1 are predominantly flanked by 5&#x2019;C and 3&#x2019;G nucleotides, implying that the CAGCTG sequence context correlates with reduced SHM of AGCT motifs.</p>
</sec>
<sec id="s3_3">
<title>CAGCTG is an E-box binding motif and E2A associates with V regions</title>
<p>CAGCTG corresponds to the CANNTG E-box binding motif of the basic helix-loop-helix TF family, which includes E2A (<xref ref-type="bibr" rid="B50">50</xref>). To predict binding probabilities of E2A to AGCT 15-mers, we used MOODS, a TF-binding prediction package which utilizes counts matrices obtained from empirical SELEX data, wherein higher MOODS scores reflect a stronger sequence match to a particular TF binding motif (<xref ref-type="bibr" rid="B62">62</xref>).</p>
<p>We found significant negative Pearson correlations of r=-0.64 and r=-0.65 between the MOODS binding scores for E2A motifs and the mutation frequencies of the central Gs (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5A</bold>
</xref>) and central Cs (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5B</bold>
</xref>) in the AGCT sites, respectively. Similar analysis for TFAP4, another E-box TF commonly expressed in B cells, showed weaker correlations (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;1</bold>
</xref>). The predicted MOODS scores for the AGCT 15-mers fell roughly into three discrete tiers. Tier 1, having the highest MOODS scores but generally lower mutation frequencies, and consisting almost entirely of CAGCTG-containing 15-mers (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5A, B</bold>
</xref>). Tier 2, having intermediate MOODS scores with a wide range of mutation frequencies. Importantly, although this tier consists of a mixture of CAGCTG and non-CAGCTG 15-mers, the former showed a tendency to be less mutated than the latter (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5A, B</bold>
</xref>). Interestingly, the FW2 CAGCTG 15-mers observed to be highly mutating in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> fall into this tier and contain central Gs (green crosses in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5A</bold>
</xref>). This indicates lower fidelity to the E2A motif than the lowly mutating FW1 CAGCTGs and may explain, in part, the higher mutation frequency due to diminished E2A binding. Tier 3, which harbored the lowest MOODS scores and generally higher mutation frequencies, consisted mostly of non-CAGCTG 15-mers (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5A, B</bold>
</xref>). These results suggest that the binding probability of E2A to an AGCT-centered 15-mer negatively correlates with mutation frequency of that AGCT.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Scatter plots depicting the correlation between observed mutation frequencies and E2A MOODS scores for AGCT 15-mers. <bold>(A, B)</bold> analysis of 15mers centered at the central G <bold>(A)</bold> or central C <bold>(B)</bold>. Each point represents a 15-mer and is colored by IMGT subregion, with CAGCTG 15-mers indicated with a cross (x). The red lines indicate the best fit with intercept and coefficient computed using a linear regression. The r value is the Pearson correlation coefficient, and the <italic>P</italic> value is computed using a Wald test. The three tiers (Tier 1&#x2013;3) that the MOODs scores fall into are labeled.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1407470-g005.tif"/>
</fig>
<p>Next, we determined the distribution of all potential E2A sites across all human IGHV genes. We segmented germline sequences for the 220 alleles obtained from the IMGT database into six subregions and counted the number of occurrences of CAGCTG (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>) and the more general CANNTG (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>) E-box motifs. CAGCTG motifs were mostly distributed in the FW1 region, with the IGHV2 family notably lacking them (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>). The IGHV4 family has the highest density of CANNTG motifs while most of IGHV2 family have E-box motifs in the FW3 region (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>). Overall, each of the 220 alleles had at least one E-box motif, with most of them in FW1 and/or the leader-intron-leader (L-intron-L) sequence which immediately precedes FW1 (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>).</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Heatmap depicting the number of <bold>(A)</bold> CAGCTG and <bold>(B)</bold> CANNTG E-box motifs in human germline IGHV genes (y axis) classified into subregions (x axis) based on the IMGT nomenclature. Each cell corresponds to a distinct IGHV sub-region and is colored by the number of E-box motifs (between 0 and 3) in that sub-region as shown in the key on the right. Each row corresponds to a unique IGHV allele. The dashed horizontal lines represent boundaries between the seven IGHV families (IGHV1&#x2013;7).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1407470-g006.tif"/>
</fig>
<p>To determine whether E2A associates with IGHV regions, we analyzed E2A ChIP-seq datasets derived from Ramos (<xref ref-type="bibr" rid="B45">45</xref>) and GM12878 (<xref ref-type="bibr" rid="B64">64</xref>) B cell lines. After aligning these data to the hg38 reference genome using bowtie2 (<xref ref-type="bibr" rid="B65">65</xref>), we used MACS2 (<xref ref-type="bibr" rid="B67">67</xref>) to call peaks and then conducted a motif enrichment analysis using HOMER (<xref ref-type="bibr" rid="B68">68</xref>). We saw that for both Ramos (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;1</bold>
</xref>) and GM12878 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;2</bold>
</xref>) cell lines, the CAGCTG motif corresponding to the E2A TF binding motif was highly enriched and among the top two most significant results. We calculated the reads per kilobase million (RPKM) values of 500bp bins in both the E2A ChIP-seq and the IgG control ChIP-seq alignments and plotted their correlations. In both Ramos (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7A</bold>
</xref>) and GM12878 (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7B</bold>
</xref>) cells, the bins containing the rearranged V region (IGHV4&#x2013;34 in Ramos (<xref ref-type="bibr" rid="B71">71</xref>) and IGHV3&#x2013;21 in GM12878) (<xref ref-type="bibr" rid="B72">72</xref>)) were enriched for E2A binding, as was the bin containing the <italic>IGH</italic> E&#x3bc; enhancer, which serves as a positive control for E2A binding (<xref ref-type="bibr" rid="B43">43</xref>). However, bins containing a negative control region, TRBV20&#x2013;1, a commonly used T-cell receptor V gene (<xref ref-type="bibr" rid="B73">73</xref>), showed no enrichment for E2A binding in either cell line (<xref ref-type="fig" rid="f7">
<bold>Figures&#xa0;7A, B</bold>
</xref>). Thus, E2A can directly associate with V regions.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>
<bold>(A, B)</bold> Scatter plots depicting correlations between IgG control and E2A ChIP-seq shown as reads per kilobase million (RPKM) values in 500 bp genomic bins for Ramos <bold>(A)</bold> and GM12878 <bold>(B)</bold> cells. Bins containing the rearranged IGHV, E&#x3bc; enhancer and TRBV20&#x2013;1 are highlighted in blue, orange and green, respectively. The black line represents the y=x diagonal.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1407470-g007.tif"/>
</fig>
<p>Altogether, these results suggest that E2A association with CAGCTG motifs in FW1 suppress AID targeting to these AGCTs, thereby providing a plausible explanation for the strong negative correlation between this motif context and mutation frequency.</p>
</sec>
<sec id="s3_4">
<title>E-box and PyPy dimers contribute independently to AGCT mutability</title>
<p>We additionally sought to compare the role of E2A binding with another sequence-level determinant of AGCT mutability proposed in a recent study (<xref ref-type="bibr" rid="B19">19</xref>), namely, the presence of PyPy dimers. In this study, a higher frequency of PyPy dimers in the 6 nucleotides 5&#x2019; to the AGCT was associated with increased mutation frequency of the central C residue (<xref ref-type="bibr" rid="B19">19</xref>). Therefore, we counted the number of PyPy dimers in the 6 nt region immediately upstream of AGCT motifs in our k-mer dataset and fit a linear regression model predicting mutation frequency, including an indicator variable for the presence of the E-box motif. This binary E-box indicator variable had a Pearson correlation of r=-0.61 with the mutation frequency, while the integer PyPy count had a correlation of 0.51 (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). Importantly, therefore, both variables individually correlate with mutation frequency in directions consistent with our expectations, that is, positive for PyPy counts, which increases mutation frequency, and negative for the presence of an E-box, which reduces mutation frequency.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Pearson r and R<sup>2</sup> for correlations and model performance against mutation frequency.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Variable</th>
<th valign="top" align="center">Pearson r</th>
<th valign="top" align="center">R<sup>2</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">PyPy Counts</td>
<td valign="top" align="center">0.51</td>
<td valign="top" align="center">0.25</td>
</tr>
<tr>
<td valign="top" align="left">E-box indicator</td>
<td valign="top" align="center">-0.61</td>
<td valign="top" align="center">0.39</td>
</tr>
<tr>
<td valign="top" align="left">PyPy + E-box</td>
<td valign="top" align="center">N/A</td>
<td valign="top" align="center">0.45</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Given this trend, we expected a linear model with both variables to have a higher performance than a model with either individual variable, as measured by R<sup>2</sup>, which directly reflects the proportion of variance in the output variable (mutation frequency) explained by the input variables (E-box motif, PyPy richness, or both). The combined regression model achieved an R<sup>2</sup> of 0.45 meaning that 45% of the variance in mutation frequency is explained by the presence of E-box and PyPy motifs (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). The regression model with only the PyPy counts variable achieved an R<sup>2</sup> of 0.25 while the model with only the E-box indicator variable achieved a higher, and closer to the combined, R<sup>2</sup> value of 0.39 (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). Of note, the coefficients generated by the model had signs appropriate to the direction of correlation with mutation frequency, that is, positive for PyPy counts and negative for the E-box indicator (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>).</p>
<p>These results lead to the conclusion that both mechanisms, decreasing mutation frequency of FW1 AGCTs, plausibly through E2A binding, and increasing mutation frequency of AGCTs through increased AID binding to flexible PyPy-rich DNA can contribute independently to the observed mutability. Importantly, however, the R<sup>2</sup> values observed from these analyses also imply that additional mechanisms are necessary to fully explain AGCT mutability.</p>
</sec>
<sec id="s3_5">
<title>Ablation of the CAGCTG motif is associated with a significant increase in mutation frequency</title>
<p>To better understand the relationship between the E-box motif and the mutability of the central nucleotides, we examined mutations of the CAGCTG hotspot in FW1. We hypothesized that if this motif context negatively contributes to SHM, then naturally occurring mutations that ablate this context would be expected to increase mutation frequency of the AGCT within it.</p>
<p>Due to the paucity of non-productive sequences in our dataset, we examined synonymous mutations (i.e. those that do not cause changes in protein sequence) in productive BCR sequences to preclude any effect of affinity selection. Specifically, we focused on the G residues at positions 3 (G<sub>3</sub>) and 6 (G<sub>6</sub>) of CAGCTG. Importantly, the FW1 CAGCTG motif occurs at position 7&#x2013;12 of the V segment and is always in frame, such that mutations at G<sub>3</sub> and G<sub>6</sub> are in the third position of their respective codons. Thus, G<sub>3</sub>&gt;A<sub>3</sub> mutations are synonymous since CAG and CAA are degenerate codons for glutamine. Similarly, G<sub>6</sub>&gt;H<sub>6</sub> mutations (where H = A/C/T) are also synonymous since CTG, CTA, CTC and CTT are degenerate codons for leucine. Importantly, G<sub>6</sub>&gt;H<sub>6</sub> mutations (CAGCTH) would ablate the E-box motif. Thus, we compared mutation frequency at the central AGCT in clonal groups having an unmutated CAGCTG in FW1 or a CAGCTH in the same position.</p>
<p>To prevent double counting of mutations occurring during clonal expansion, we selected a sequence at random from each clonal group (<xref ref-type="bibr" rid="B57">57</xref>). Our sequence data consisted of 642,367 clonal groups, of which 504,333 had a sequence identity in position 7&#x2013;12 of the V segment, corresponding to one of the four motifs of interest: unmutated CAGCTG (G<sub>3</sub>/G<sub>6</sub>), single mutant CAACTG (A<sub>3</sub>/G<sub>6</sub>), single mutant CAGCTH (G<sub>3</sub>/H<sub>6</sub>) and double mutant CAACTH (A<sub>3</sub>/H<sub>6</sub>). We counted the number of occurrences of each (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>) and used these numbers to calculate mutation frequencies for sites 3 and 6 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;3</bold>
</xref>). We observed that the mutation frequency of G<sub>3</sub> increases from 9.6% when G<sub>6</sub> is unmutated to 21.5% when G<sub>6</sub> is mutated, a highly significant difference (Fisher test, <italic>P</italic>&lt;10<sup>&#x2013;16</sup>) (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>). Thus, the presence of an intact CAGCTG motif is strongly associated with lower mutation of the AGCT within it.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Counts of synonymous mutations at sites 3 and 6 of the CAGCTG motif across clones.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Site 6/Site 3</th>
<th valign="top" align="center">Unmutated (G<sub>3</sub>)</th>
<th valign="top" align="center">Mutated (A<sub>3</sub>)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Unmutated (G<sub>6</sub>)</td>
<td valign="top" align="center">426962</td>
<td valign="top" align="center">45537</td>
</tr>
<tr>
<td valign="top" align="left">Mutated (H<sub>6</sub>)</td>
<td valign="top" align="center">24988</td>
<td valign="top" align="center">6846</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Flowchart depicting synonymous mutations of the Gs at site 3 (G<sub>3</sub>) of the CAGCTG E-box motif in FW1. Calculated mutation frequencies of G<sub>3</sub> sites before and after mutation of G<sub>6</sub> are indicated beside the arrows. Calculations were conducted using productive, clonally independent sequences across clones.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1407470-g008.tif"/>
</fig>
<p>Altogether, our results support the notion that FW1 AGCT motifs occurring in the context of the E-box binding motif, CAGCTG, lead to dampened SHM in these locales.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>In this study, we use interpretable deep learning to provide evidence for the role of DNA sequence context in negatively modulating SHM at AGCT motifs located at the 5&#x2019; end of most human IGHV genes, except those of the VH2 family. Our work suggests that the occurrence of this AGCT in the context of a CAGCTG E-box motif correlates strongly with reduced SHM. Together with the fact that E2A can associate with VH4&#x2013;34 and VH3&#x2013;21, we propose that SHM may be dampened at these motifs, at least in part, by the association of E-box-binding TFs. The decrease in AID mediated mutations could occur through a variety of mechanisms including changes in transcription elongation or pausing, or a decrease in the recruitment of AID or its associated cofactors. In effect, this would constitute a new, suppressive mechanism contributing to the differential mutability of AGCT motifs in specific contexts. Our work, therefore, provides a conceptual framework to guide further studies aimed at identifying similar mechanisms regulating local SHM probabilities at other WRCH motifs, including other AGCT motif contexts, perhaps involving different TFs or combinations thereof.</p>
<p>How might E2A binding to CAGCTG suppress SHM? E2A binds ssDNA <italic>in vitro</italic> and has a higher affinity for CAGCTG than for the canonical dsDNA binding site, CAGGTG (<xref ref-type="bibr" rid="B74">74</xref>). Additionally, mutations in the middle nucleotides of the CANNTG motif reduced E2A binding to ssDNA substantially, but not to dsDNA (<xref ref-type="bibr" rid="B74">74</xref>). These results, along with our analyses, suggest a competitive binding model for the significantly weaker mutability of the FW1 AGCT motifs wherein E2A binding to ssDNA may prevent AID from accessing exposed CAGCTG motifs. Since the CAGCTG E-box TF binding motif is palindromic, E2A could potentially access both strands, for instance, under conditions of transcription-induced negative supercoiling where both template and non-template strands can acquire transient ssDNA states (<xref ref-type="bibr" rid="B75">75</xref>). If so, E2A could restrict AID from accessing CAGCTG-containing ssDNA on either strand. Since the processing of SHM-induced mismatches in the V region can result in DNA double-strand breaks (<xref ref-type="bibr" rid="B76">76</xref>), we expect that such a mechanism would also impact on the formation of these lesions.</p>
<p>Collectively, these findings raise two hypotheses that merit further investigation. Firstly, other E-box-binding TFs expressed in B cells may associate with CAGCTG in FW1 and contribute to suppressing SHM. Secondly, AID accessibility at other WRCH motifs may be subject to similar negative regulation mediated by the competitive binding of different TFs. Such analyses are also necessary at non-IG SHM target loci implicated in B lymphomagenesis, such as <italic>MYC</italic> and <italic>BCL6</italic>, to ask whether similar mechanisms regulate differential mutability during off-target SHM.</p>
<p>Our analysis of PyPy richness revealed a positive correlation of this feature with AGCT mutability, in agreement with the <italic>in vitro</italic> findings of Wang et&#xa0;al. (<xref ref-type="bibr" rid="B19">19</xref>). Our results also suggest that PyPy-richness and E-box motifs can work independently in determining the mutability of AGCT motifs. Thus, we conclude that SHM enhancement via increased ssDNA flexibility conferred by PyPy motifs and SHM suppression via E2A binding at E-box motifs constitute two distinct mechanisms to achieve differential mutability of AGCT motifs. Importantly, however, it is evident from our data that these features, either singly or in combination, cannot fully explain AGCT mutability, implying that additional as-yet-unknown mechanisms exist that contribute to differential mutability, such as the influence of position, as in the case of some lowly mutating FW1 AGCTs that do not lie in a CAGCTG context (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>).</p>
<p>In conclusion, our study reveals the complexity underlying local AID targeting and argues that the eventual discrete SHM profiles result from multiple mechanisms that either strengthen or dampen SHM. As exemplified by our study, deep learning tools will be an important resource for mining mutational datasets to gain further insights into these mechanisms.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. All code use in this study is available at  <uri xlink:href="https://github.com/abhikt/e2a_paper">https://github.com/abhikt/e2a_paper</uri>. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>AT: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing, Data curation, Formal analysis, Investigation, Methodology, Validation, Visualization. TM: Writing &#x2013; original draft, Conceptualization, Funding acquisition, Project administration, Resources, Supervision. RP: Supervision, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by grant NIH R01AI132507 to TM. The IMP is core funded by Boehringer Ingelheim. The funders had no role in study design, data collection, and interpretation or the decision to submit the work for publication.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We dedicate this paper to TM, initially the corresponding author and supervisor of this study, who tragically passed away on November 3<sup>rd</sup>, 2023. We would also like to thank Matthew Scharff (Albert Einstein College of Medicine, New York), Ursula Sch&#xf6;berl (IMP, Vienna), Johanna Fitz (IMP, Vienna), and Ramana Davuluri (Stony Brook University, New York) for critical reading of the manuscript and helpful suggestions.</p>
</ack>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fimmu.2024.1407470/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fimmu.2024.1407470/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Image_1.jpeg" id="SF1" mimetype="image/jpeg"/>
<supplementary-material xlink:href="Table_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rajewsky</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>Clonal selection and learning in the antibody system</article-title>. <source>Nature</source>. (<year>1996</year>) <volume>381</volume>:<page-range>751&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/381751a0</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Victora</surname> <given-names>GD</given-names>
</name>
<name>
<surname>Nussenzweig</surname> <given-names>MC</given-names>
</name>
</person-group>. <article-title>Germinal centers</article-title>. <source>Annu Rev Immunol</source>. (<year>2022</year>) <volume>40</volume>:<page-range>413&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev-immunol-120419&#x2013;022408</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muramatsu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kinoshita</surname> <given-names>K</given-names>
</name>
<name>
<surname>Fagarasan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Yamada</surname> <given-names>S</given-names>
</name>
<name>
<surname>Shinkai</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Honjo</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>Class switch recombination and hypermutation require activation-induced cytidine deaminase (AID), a potential RNA editing enzyme</article-title>. <source>Cell</source>. (<year>2000</year>) <volume>102</volume>:<page-range>553&#x2013;63</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0092&#x2013;8674(00)00078&#x2013;7</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Revy</surname> <given-names>P</given-names>
</name>
<name>
<surname>Muto</surname> <given-names>T</given-names>
</name>
<name>
<surname>Levy</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Geissmann</surname> <given-names>F</given-names>
</name>
<name>
<surname>Plebani</surname> <given-names>A</given-names>
</name>
<name>
<surname>Sanal</surname> <given-names>O</given-names>
</name>
<etal/>
</person-group>. <article-title>Activation-induced cytidine deaminase (AID) deficiency causes the autosomal recessive form of the hyper-igM syndrome (HIGM2)</article-title>. <source>Cell</source>. (<year>2000</year>) <volume>102</volume>:<page-range>565&#x2013;75</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0092&#x2013;8674(00)00079&#x2013;9</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Petersen-Mahrt</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Harris</surname> <given-names>RS</given-names>
</name>
<name>
<surname>Neuberger</surname> <given-names>MS</given-names>
</name>
</person-group>. <article-title>AID mutates E. coli suggesting a DNA deamination mechanism for antibody diversification</article-title>. <source>Nature</source>. (<year>2002</year>) <volume>418</volume>:<fpage>99</fpage>&#x2013;<lpage>103</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature00862</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramiro</surname> <given-names>AR</given-names>
</name>
<name>
<surname>Stavropoulos</surname> <given-names>P</given-names>
</name>
<name>
<surname>Jankovic</surname> <given-names>M</given-names>
</name>
<name>
<surname>Nussenzweig</surname> <given-names>MC</given-names>
</name>
</person-group>. <article-title>Transcription enhances AID-mediated cytidine deamination by exposing single-stranded DNA on the nontemplate strand</article-title>. <source>Nat Immunol</source>. (<year>2003</year>) <volume>4</volume>:<page-range>452&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/NI920</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bransteitter</surname> <given-names>R</given-names>
</name>
<name>
<surname>Pham</surname> <given-names>P</given-names>
</name>
<name>
<surname>Scharff</surname> <given-names>MD</given-names>
</name>
<name>
<surname>Goodman</surname> <given-names>MF</given-names>
</name>
</person-group>. <article-title>Activation-induced cytidine deaminase deaminates deoxycytidine on single-stranded DNA but requires the action of RNase</article-title>. <source>Proc Natl Acad Sci USA</source>. (<year>2003</year>) <volume>100</volume>:<page-range>4102&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.0730835100</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pham</surname> <given-names>P</given-names>
</name>
<name>
<surname>Bransteitter</surname> <given-names>R</given-names>
</name>
<name>
<surname>Petruska</surname> <given-names>J</given-names>
</name>
<name>
<surname>Goodman</surname> <given-names>MF</given-names>
</name>
</person-group>. <article-title>Processive AID-catalysed cytosine deamination on single-stranded DNA simulates somatic hypermutation</article-title>. <source>Nature</source>. (<year>2003</year>) <volume>424</volume>:<page-range>103&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature01760</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rogozin</surname> <given-names>IB</given-names>
</name>
<name>
<surname>Diaz</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Cutting edge: DGYW/WRCH is a better predictor of mutability at G:C bases in ig hypermutation than the widely accepted RGYW/WRCY motif and probably reflects a two-step activation-induced cytidine deaminase-triggered process</article-title>. <source>J Immunol</source>. (<year>2004</year>) <volume>172</volume>:<page-range>3382&#x2013;4</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.172.6.3382</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peled</surname> <given-names>JU</given-names>
</name>
<name>
<surname>Kuang</surname> <given-names>FL</given-names>
</name>
<name>
<surname>Iglesias-Ussel</surname> <given-names>MD</given-names>
</name>
<name>
<surname>Roa</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kalis</surname> <given-names>SL</given-names>
</name>
<name>
<surname>Goodman</surname> <given-names>MF</given-names>
</name>
<etal/>
</person-group>. <article-title>The biochemistry of somatic hypermutation</article-title>. <source>Annu Rev Immunol</source>. (<year>2008</year>) <volume>26</volume>:<fpage>481</fpage>&#x2013;<lpage>511</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev.immunol.26.021607.090236</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Di Noia</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Neuberger</surname> <given-names>MS</given-names>
</name>
</person-group>. <article-title>Molecular mechanisms of antibody somatic hypermutation</article-title>. <source>Annu Rev Biochem</source>. (<year>2007</year>) <volume>76</volume>:<fpage>1</fpage>&#x2013;<lpage>22</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev.biochem.76.061705.090740</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Seija</surname> <given-names>N</given-names>
</name>
<name>
<surname>Di Noia</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Martin</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>AID in antibody diversification: there and back again</article-title>. <source>Trends Immunol</source>. (<year>2020</year>) <volume>41</volume>:<fpage>586</fpage>&#x2013;<lpage>600</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.it.2020.04.009</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Methot</surname> <given-names>SP</given-names>
</name>
<name>
<surname>Di Noia</surname> <given-names>JM</given-names>
</name>
</person-group>. <article-title>Molecular mechanisms of somatic hypermutation and class switch recombination</article-title>. <source>Adv Immunol</source>. (<year>2017</year>) <volume>133</volume>:<fpage>37</fpage>&#x2013;<lpage>87</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/bs.ai.2016.11.002</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname> <given-names>L</given-names>
</name>
<name>
<surname>Chahwan</surname> <given-names>R</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Pham</surname> <given-names>PT</given-names>
</name>
<name>
<surname>Goodman</surname> <given-names>MF</given-names>
</name>
<etal/>
</person-group>. <article-title>Overlapping hotspots in CDRs are critical sites for V region diversification</article-title>. <source>Proc Natl Acad Sci</source>. (<year>2015</year>) <volume>112</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1500788112</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>JQ</given-names>
</name>
<name>
<surname>Kleinstein</surname> <given-names>SH</given-names>
</name>
</person-group>. <article-title>Position-dependent differential targeting of somatic hypermutation</article-title>. <source>J Immunol</source>. (<year>2020</year>) <volume>205</volume>:<page-range>3468&#x2013;79</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.2000496</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spisak</surname> <given-names>N</given-names>
</name>
<name>
<surname>Walczak</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Mora</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>Learning the heterogeneous hypermutation landscape of immunoglobulins from high-throughput repertoire data</article-title>. <source>Nucleic Acids Res</source>. (<year>2020</year>) <volume>48</volume>:<page-range>10702&#x2013;12</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkaa825</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pham</surname> <given-names>P</given-names>
</name>
<name>
<surname>Calabrese</surname> <given-names>P</given-names>
</name>
<name>
<surname>Park</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Goodman</surname> <given-names>MF</given-names>
</name>
</person-group>. <article-title>Analysis of a single-stranded DNA-scanning process in which activation-induced deoxycytidine deaminase (AID) deaminates C to U haphazardly and inefficiently to ensure mutational diversity</article-title>. <source>J Biol Chem</source>. (<year>2011</year>) <volume>286</volume>:<page-range>24931&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1074/jbc.M111.241208</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Krantsevich</surname> <given-names>A</given-names>
</name>
<name>
<surname>MacCarthy</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>Deep learning model of somatic hypermutation reveals importance of sequence context beyond hotspot targeting</article-title>. <source>iScience</source>. (<year>2022</year>) <volume>25</volume>:<elocation-id>103668</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.isci.2021.103668</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Hwang</surname> <given-names>JK</given-names>
</name>
<name>
<surname>Zhan</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lian</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Mesoscale DNA feature in antibody-coding sequence facilitates somatic hypermutation</article-title>. <source>Cell</source>. (<year>2023</year>) <volume>186</volume>:<fpage>2193</fpage>&#x2013;<lpage>2207.e19</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cell.2023.03.030</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duan</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Baughn</surname> <given-names>LB</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>V</given-names>
</name>
<name>
<surname>MacCarthy</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>Role of Dot1L and H3K79 methylation in regulating somatic hypermutation of immunoglobulin genes</article-title>. <source>Proc Natl Acad Sci</source>. (<year>2021</year>) <volume>118</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.2104013118</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Begum</surname> <given-names>NA</given-names>
</name>
<name>
<surname>Stanlie</surname> <given-names>A</given-names>
</name>
<name>
<surname>Nakata</surname> <given-names>M</given-names>
</name>
<name>
<surname>Akiyama</surname> <given-names>H</given-names>
</name>
<name>
<surname>Honjo</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>The histone chaperone spt6 is required for activation-induced cytidine deaminase target determination through H3K4me3 regulation</article-title>. <source>J Biol Chem</source>. (<year>2012</year>) <volume>287</volume>:<page-range>32415&#x2013;29</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1074/jbc.M112.351569</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>G</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>V</given-names>
</name>
<name>
<surname>MacCarthy</surname> <given-names>T</given-names>
</name>
<name>
<surname>Scharff</surname> <given-names>MD</given-names>
</name>
</person-group>. <article-title>HIRA-dependent H3.3 deposition and its modification facilitate somatic hypermutation of immunoglobulin gene by maintaining the proper chromatin context and transcription</article-title>. <source>J Immunol</source>. (<year>2021</year>) <volume>206</volume>:<page-range>63.04&#x2013;4</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.206.supp.63.04</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aida</surname> <given-names>M</given-names>
</name>
<name>
<surname>Hamad</surname> <given-names>N</given-names>
</name>
<name>
<surname>Stanlie</surname> <given-names>A</given-names>
</name>
<name>
<surname>Begum</surname> <given-names>NA</given-names>
</name>
<name>
<surname>Honjo</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>Accumulation of the FACT complex, as well as histone H3.3, serves as a target marker for somatic hypermutation</article-title>. <source>Proc Natl Acad Sci USA</source>. (<year>2013</year>) <volume>110</volume>:<page-range>7784&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1305859110</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jeevan-Raj</surname> <given-names>BP</given-names>
</name>
<name>
<surname>Robert</surname> <given-names>I</given-names>
</name>
<name>
<surname>Heyer</surname> <given-names>V</given-names>
</name>
<name>
<surname>Page</surname> <given-names>A</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>JH</given-names>
</name>
<name>
<surname>Cammas</surname> <given-names>F</given-names>
</name>
<etal/>
</person-group>. <article-title>Epigenetic tethering of AID to the donor switch region during immunoglobulin class switch recombination</article-title>. <source>J Exp Med</source>. (<year>2011</year>) <volume>208</volume>:<page-range>1649&#x2013;60</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1084/jem.20110118</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stanlie</surname> <given-names>A</given-names>
</name>
<name>
<surname>Aida</surname> <given-names>M</given-names>
</name>
<name>
<surname>Muramatsu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Honjo</surname> <given-names>T</given-names>
</name>
<name>
<surname>Begum</surname> <given-names>NA</given-names>
</name>
</person-group>. <article-title>Histone3 lysine4 trimethylation regulated by the facilitates chromatin transcription complex is critical for DNA cleavage in class switch recombination</article-title>. <source>Proc Natl Acad Sci USA</source>. (<year>2010</year>) <volume>107</volume>:<page-range>22190&#x2013;5</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1016923108</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bradley</surname> <given-names>SP</given-names>
</name>
<name>
<surname>Kaminski</surname> <given-names>DA</given-names>
</name>
<name>
<surname>Peters</surname> <given-names>AHFM</given-names>
</name>
<name>
<surname>Jenuwein</surname> <given-names>T</given-names>
</name>
<name>
<surname>Stavnezer</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>The histone methyltransferase suv39h1 increases class switch recombination specifically to igA</article-title>. <source>J Immunol</source>. (<year>2006</year>) <volume>177</volume>:<page-range>1179&#x2013;88</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.177.2.1179</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuang</surname> <given-names>FL</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Scharff</surname> <given-names>MD</given-names>
</name>
</person-group>. <article-title>H3 trimethyl K9 and H3 acetyl K9 chromatin modifications are associated with class switch recombination</article-title>. <source>Proc Natl Acad Sci USA</source>. (<year>2009</year>) <volume>106</volume>:<page-range>5288&#x2013;93</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/PNAS.0901368106</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Daniel</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Santos</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Zang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Schwab</surname> <given-names>KR</given-names>
</name>
<name>
<surname>Jankovic</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>PTIP promotes chromatin changes critical for immunoglobulin class switch recombination</article-title>. <source>Science</source>. (<year>2010</year>) <volume>329</volume>:<page-range>917&#x2013;23</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.1187942</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaidyanathan</surname> <given-names>B</given-names>
</name>
<name>
<surname>Chaudhuri</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Epigenetic codes programming class switch recombination</article-title>. <source>Front Immunol</source>. (<year>2015</year>) <volume>6</volume>:<elocation-id>405/PDF</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2015.00405</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pavri</surname> <given-names>R</given-names>
</name>
<name>
<surname>Gazumyan</surname> <given-names>A</given-names>
</name>
<name>
<surname>Jankovic</surname> <given-names>M</given-names>
</name>
<name>
<surname>Virgilio</surname> <given-names>M</given-names>
</name>
<name>
<surname>Klein</surname> <given-names>I</given-names>
</name>
</person-group>. <article-title>Activation-induced cytidine deaminase targets DNA at sites of RNA polymerase II stalling by interaction with Spt5</article-title>. <source>Cell</source>. (<year>2010</year>) <volume>143</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cell.2010.09.017</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#xc1;lvarez-Prado</surname> <given-names>&#xc1;F</given-names>
</name>
<name>
<surname>P&#xe9;rez-Dur&#xe1;n</surname> <given-names>P</given-names>
</name>
<name>
<surname>P&#xe9;rez-Garc&#xed;a</surname> <given-names>A</given-names>
</name>
<name>
<surname>Benguria</surname> <given-names>A</given-names>
</name>
<name>
<surname>Torroja</surname> <given-names>C</given-names>
</name>
<name>
<surname>de Y&#xe9;benes</surname> <given-names>VG</given-names>
</name>
<etal/>
</person-group>. <article-title>A broad atlas of somatic hypermutation allows prediction of activation-induced deaminase targets</article-title>. <source>J Exp Med</source>. (<year>2018</year>) <volume>215</volume>:<page-range>761&#x2013;71</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1084/jem.20171738</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rajagopal</surname> <given-names>D</given-names>
</name>
<name>
<surname>Maul</surname> <given-names>RW</given-names>
</name>
<name>
<surname>Ghosh</surname> <given-names>A</given-names>
</name>
<name>
<surname>Chakraborty</surname> <given-names>T</given-names>
</name>
<name>
<surname>Khamlichi</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Sen</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Immunoglobulin switch mu sequence causes RNA polymerase II accumulation and reduces dA hypermutation</article-title>. <source>J Exp Med</source>. (<year>2009</year>) <volume>206</volume>:<page-range>1237&#x2013;44</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1084/JEM.20082514</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>L</given-names>
</name>
<name>
<surname>Wuerffel</surname> <given-names>R</given-names>
</name>
<name>
<surname>Feldman</surname> <given-names>S</given-names>
</name>
<name>
<surname>Khamlichi</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Kenter</surname> <given-names>AL</given-names>
</name>
</person-group>. <article-title>S region sequence, RNA polymerase II, and histone modifications create chromatin accessibility during class switch recombination</article-title>. <source>J Exp Med</source>. (<year>2009</year>) <volume>206</volume>:<page-range>1817&#x2013;30</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1084/JEM.20081678</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maul</surname> <given-names>RW</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Venkataraman</surname> <given-names>L</given-names>
</name>
<name>
<surname>Giorgetti</surname> <given-names>CA</given-names>
</name>
<name>
<surname>Press</surname> <given-names>JL</given-names>
</name>
<name>
<surname>Denizot</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Spt5 accumulation at variable genes distinguishes somatic hypermutation in germinal center B cells from ex vivo&#x2013;activated cells</article-title>. <source>J Exp Med</source>. (<year>2014</year>) <volume>211</volume>:<page-range>2297&#x2013;306</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1084/jem.20131512</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tarsalainen</surname> <given-names>A</given-names>
</name>
<name>
<surname>Maman</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Meng</surname> <given-names>F-L</given-names>
</name>
<name>
<surname>Kyl&#xe4;niemi</surname> <given-names>MK</given-names>
</name>
<name>
<surname>Soikkeli</surname> <given-names>A</given-names>
</name>
<name>
<surname>Budzy&#x144;ska</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Ig enhancers increase RNA polymerase II stalling at somatic hypermutation target sequences</article-title>. <source>J Immunol</source>. (<year>2022</year>) <volume>208</volume>:<page-range>143&#x2013;54</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.2100923</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Canugovi</surname> <given-names>C</given-names>
</name>
<name>
<surname>Samaranayake</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bhagwat</surname> <given-names>AS</given-names>
</name>
</person-group>. <article-title>Transcriptional pausing and stalling causes multiple clustered mutations by human activation-induced deaminase</article-title>. <source>FASEB J</source>. (<year>2009</year>) <volume>23</volume>:<fpage>34</fpage>&#x2013;<lpage>44</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1096/fj.08&#x2013;115352</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Basu</surname> <given-names>U</given-names>
</name>
<name>
<surname>Meng</surname> <given-names>F-L</given-names>
</name>
<name>
<surname>Keim</surname> <given-names>C</given-names>
</name>
<name>
<surname>Grinstein</surname> <given-names>V</given-names>
</name>
<name>
<surname>Pefanis</surname> <given-names>E</given-names>
</name>
<name>
<surname>Eccleston</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>The RNA exosome targets the AID cytidine deaminase to both strands of transcribed duplex DNA substrates</article-title>. <source>Cell</source>. (<year>2011</year>) <volume>144</volume>:<page-range>353&#x2013;63</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cell.2011.01.001</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pefanis</surname> <given-names>E</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Rothschild</surname> <given-names>G</given-names>
</name>
<name>
<surname>Lim</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chao</surname> <given-names>J</given-names>
</name>
<name>
<surname>Rabadan</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Noncoding RNA transcription targets AID to divergently transcribed loci in B cells</article-title>. <source>Nature</source>. (<year>2014</year>) <volume>514</volume>:<page-range>389&#x2013;93</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature13580</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meng</surname> <given-names>F-L</given-names>
</name>
<name>
<surname>Du</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Federation</surname> <given-names>A</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Kieffer-Kwon</surname> <given-names>K-R</given-names>
</name>
<etal/>
</person-group>. <article-title>Convergent transcription at intragenic super-enhancers targets AID-initiated genomic instability</article-title>. <source>Cell</source>. (<year>2014</year>) <volume>159</volume>:<page-range>1538&#x2013;48</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cell.2014.11.014</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schoeberl</surname> <given-names>UE</given-names>
</name>
<name>
<surname>Fitz</surname> <given-names>J</given-names>
</name>
<name>
<surname>Froussios</surname> <given-names>K</given-names>
</name>
<name>
<surname>Valieris</surname> <given-names>R</given-names>
</name>
<name>
<surname>Ourailidis</surname> <given-names>I</given-names>
</name>
<name>
<surname>Makharova</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Somatic hypermutation patterns in immunoglobulin variable regions are established independently of the local transcriptional landscape</article-title>. <source>bioRxiv</source>. (<year>2023</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2022.05.21.492925</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kohler</surname> <given-names>KM</given-names>
</name>
<name>
<surname>McDonald</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Duke</surname> <given-names>JL</given-names>
</name>
<name>
<surname>Arakawa</surname> <given-names>H</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kleinstein</surname> <given-names>SH</given-names>
</name>
<etal/>
</person-group>. <article-title>Identification of core DNA elements that target somatic hypermutation</article-title>. <source>J Immunol</source>. (<year>2012</year>) <volume>189</volume>:<page-range>5314&#x2013;26</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.1202082</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buerstedde</surname> <given-names>J-M</given-names>
</name>
<name>
<surname>Alinikula</surname> <given-names>J</given-names>
</name>
<name>
<surname>Arakawa</surname> <given-names>H</given-names>
</name>
<name>
<surname>McDonald</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Schatz</surname> <given-names>DG</given-names>
</name>
</person-group>. <article-title>Targeting of somatic hypermutation by immunoglobulin enhancer and enhancer-like sequences</article-title>. <source>PloS Biol</source>. (<year>2014</year>) <volume>12</volume>:<elocation-id>e1001831</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pbio.1001831</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dinesh</surname> <given-names>RK</given-names>
</name>
<name>
<surname>Barnhill</surname> <given-names>B</given-names>
</name>
<name>
<surname>Ilanges</surname> <given-names>A</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Michelson</surname> <given-names>DA</given-names>
</name>
<name>
<surname>Senigl</surname> <given-names>F</given-names>
</name>
<etal/>
</person-group>. <article-title>Transcription factor binding at Ig enhancers is linked to somatic hypermutation targeting</article-title>. <source>Eur J Immunol</source>. (<year>2020</year>) <volume>50</volume>:<page-range>380&#x2013;95</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/eji.201948357</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qian</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Dose</surname> <given-names>M</given-names>
</name>
<name>
<surname>Pruett</surname> <given-names>N</given-names>
</name>
<name>
<surname>Kieffer-Kwon</surname> <given-names>K-R</given-names>
</name>
<name>
<surname>Resch</surname> <given-names>W</given-names>
</name>
<etal/>
</person-group>. <article-title>B cell super-enhancers and regulatory clusters recruit AID tumorigenic activity</article-title>. <source>Cell</source>. (<year>2014</year>) <volume>159</volume>:<page-range>1524&#x2013;37</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cell.2014.11.013</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Senigl</surname> <given-names>F</given-names>
</name>
<name>
<surname>Maman</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Dinesh</surname> <given-names>RK</given-names>
</name>
<name>
<surname>Alinikula</surname> <given-names>J</given-names>
</name>
<name>
<surname>Seth</surname> <given-names>RB</given-names>
</name>
<name>
<surname>Pecnova</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>Topologically associated domains delineate susceptibility to somatic hypermutation</article-title>. <source>Cell Rep</source>. (<year>2019</year>) <volume>29</volume>:<fpage>3902</fpage>&#x2013;<lpage>3915.e8</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.celrep.2019.11.039</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schoetz</surname> <given-names>U</given-names>
</name>
<name>
<surname>Cervelli</surname> <given-names>M</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y-D</given-names>
</name>
<name>
<surname>Fiedler</surname> <given-names>P</given-names>
</name>
<name>
<surname>Buerstedde</surname> <given-names>J-M</given-names>
</name>
</person-group>. <article-title>E2A expression stimulates ig hypermutation</article-title>. <source>J Immunol</source>. (<year>2006</year>) <volume>177</volume>:<fpage>395</fpage>&#x2013;<lpage>400</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/JIMMUNOL.177.1.395</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Duke</surname> <given-names>JL</given-names>
</name>
<name>
<surname>Richter</surname> <given-names>DJ</given-names>
</name>
<name>
<surname>Vinuesa</surname> <given-names>CG</given-names>
</name>
<name>
<surname>Goodnow</surname> <given-names>CC</given-names>
</name>
<name>
<surname>Kleinstein</surname> <given-names>SH</given-names>
</name>
<etal/>
</person-group>. <article-title>Two levels of protection for the B cell genome during somatic hypermutation</article-title>. <source>Nature</source>. (<year>2008</year>) <volume>451</volume>:<page-range>841&#x2013;5</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature06547</pub-id>
</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kwon</surname> <given-names>K</given-names>
</name>
<name>
<surname>Hutter</surname> <given-names>C</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Bilic</surname> <given-names>I</given-names>
</name>
<name>
<surname>Cobaleda</surname> <given-names>C</given-names>
</name>
<name>
<surname>Malin</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Instructive role of the transcription factor E2A in early B lymphopoiesis and germinal center B cell development</article-title>. <source>Immunity</source>. (<year>2008</year>) <volume>28</volume>:<page-range>751&#x2013;62</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.immuni.2008.04.014</pub-id>
</citation>
</ref>
<ref id="B49">
<label>49</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>W&#xf6;hner</surname> <given-names>M</given-names>
</name>
<name>
<surname>Tagoh</surname> <given-names>H</given-names>
</name>
<name>
<surname>Bilic</surname> <given-names>I</given-names>
</name>
<name>
<surname>Jaritz</surname> <given-names>M</given-names>
</name>
<name>
<surname>Poliakova</surname> <given-names>DK</given-names>
</name>
<name>
<surname>Fischer</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Molecular functions of the transcription factors E2A and E2&#x2013;2 in controlling germinal center B cell and plasma cell development</article-title>. <source>J Exp Med</source>. (<year>2016</year>) <volume>213</volume>:<fpage>1201</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1084/JEM.20152002</pub-id>
</citation>
</ref>
<ref id="B50">
<label>50</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Murre</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Helix-loop-helix proteins and lymphocyte development</article-title>. <source>Nat Immunol</source>. (<year>2005</year>) <volume>6</volume>:<page-range>1079&#x2013;86</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ni1260</pub-id>
</citation>
</ref>
<ref id="B51">
<label>51</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hauser</surname> <given-names>J</given-names>
</name>
<name>
<surname>Grundstr&#xf6;m</surname> <given-names>C</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>R</given-names>
</name>
<name>
<surname>Grundstr&#xf6;m</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>Regulated localization of an AID complex with E2A, PAX5 and IRF4 at the Igh locus</article-title>. <source>Mol Immunol</source>. (<year>2016</year>) <volume>80</volume>:<fpage>78</fpage>&#x2013;<lpage>90</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.molimm.2016.10.014</pub-id>
</citation>
</ref>
<ref id="B52">
<label>52</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grundstr&#xf6;m</surname> <given-names>C</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>A</given-names>
</name>
<name>
<surname>Priya</surname> <given-names>A</given-names>
</name>
<name>
<surname>Negi</surname> <given-names>N</given-names>
</name>
<name>
<surname>Grundstr&#xf6;m</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>ETS1 and PAX5 transcription factors recruit AID to Igh DNA</article-title>. <source>Eur J Immunol</source>. (<year>2018</year>) <volume>48</volume>:<page-range>1687&#x2013;97</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/EJI.201847625</pub-id>
</citation>
</ref>
<ref id="B53">
<label>53</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Michael</surname> <given-names>N</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>HM</given-names>
</name>
<name>
<surname>Longerich</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>N</given-names>
</name>
<name>
<surname>Longacre</surname> <given-names>A</given-names>
</name>
<name>
<surname>Storb</surname> <given-names>U</given-names>
</name>
</person-group>. <article-title>The E box motif CAGGTG enhances somatic hypermutation without enhancing transcription</article-title>. <source>Immunity</source>. (<year>2003</year>) <volume>19</volume>:<page-range>235&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S1074&#x2013;7613(03)00204&#x2013;8</pub-id>
</citation>
</ref>
<ref id="B54">
<label>54</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tanaka</surname> <given-names>A</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>HM</given-names>
</name>
<name>
<surname>Ratnam</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kodgire</surname> <given-names>P</given-names>
</name>
<name>
<surname>Storb</surname> <given-names>U</given-names>
</name>
</person-group>. <article-title>Attracting AID to targets of somatic hypermutation</article-title>. <source>J Exp Med</source>. (<year>2010</year>) <volume>207</volume>:<page-range>405&#x2013;15</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1084/jem.20090821</pub-id>
</citation>
</ref>
<ref id="B55">
<label>55</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yaari</surname> <given-names>G</given-names>
</name>
<name>
<surname>Vander Heiden</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Uduman</surname> <given-names>M</given-names>
</name>
<name>
<surname>Gadala-Maria</surname> <given-names>D</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>N</given-names>
</name>
<name>
<surname>Stern</surname> <given-names>JNH</given-names>
</name>
<etal/>
</person-group>. <article-title>Models of somatic hypermutation targeting and substitution based on synonymous mutations from high-throughput immunoglobulin sequencing data</article-title>. <source>Front Immunol</source>. (<year>2013</year>) <volume>4</volume>:<elocation-id>358</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2013.00358</pub-id>
</citation>
</ref>
<ref id="B56">
<label>56</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lefranc</surname> <given-names>MP</given-names>
</name>
<name>
<surname>Giudicelli</surname> <given-names>V</given-names>
</name>
<name>
<surname>Ginestoux</surname> <given-names>C</given-names>
</name>
<name>
<surname>Bodmer</surname> <given-names>J</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname> <given-names>W</given-names>
</name>
<name>
<surname>Bontrop</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>IMGT, the international ImMunoGeneTics database</article-title>. <source>Nucleic Acids Res</source>. (<year>1999</year>) <volume>27</volume>:<page-range>209&#x2013;12</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/NAR/27.1.209</pub-id>
</citation>
</ref>
<ref id="B57">
<label>57</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Bagnara</surname> <given-names>D</given-names>
</name>
<name>
<surname>Chiorazzi</surname> <given-names>N</given-names>
</name>
<name>
<surname>Scharff</surname> <given-names>MD</given-names>
</name>
<name>
<surname>MacCarthy</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>AID overlapping and pol&#x3b7; Hotspots are key features of evolutionary variation within the human antibody heavy chain (IGHV) genes</article-title>. <source>Front Immunol</source>. (<year>2020</year>) <volume>11</volume>:<elocation-id>788</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2020.00788</pub-id>
</citation>
</ref>
<ref id="B58">
<label>58</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gupta</surname> <given-names>NT</given-names>
</name>
<name>
<surname>Vander Heiden</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Uduman</surname> <given-names>M</given-names>
</name>
<name>
<surname>Gadala-Maria</surname> <given-names>D</given-names>
</name>
<name>
<surname>Yaari</surname> <given-names>G</given-names>
</name>
<name>
<surname>Kleinstein</surname> <given-names>SH</given-names>
</name>
</person-group>. <article-title>Change-O: A toolkit for analyzing large-scale B cell immunoglobulin repertoire sequencing data</article-title>. <source>Bioinformatics</source>. (<year>2015</year>) <volume>31</volume>:<page-range>3356&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btv359</pub-id>
</citation>
</ref>
<ref id="B59">
<label>59</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Virtanen</surname> <given-names>P</given-names>
</name>
<name>
<surname>Gommers</surname> <given-names>R</given-names>
</name>
<name>
<surname>Oliphant</surname> <given-names>TE</given-names>
</name>
<name>
<surname>Haberland</surname> <given-names>M</given-names>
</name>
<name>
<surname>Reddy</surname> <given-names>T</given-names>
</name>
<name>
<surname>Cournapeau</surname> <given-names>D</given-names>
</name>
<etal/>
</person-group>. <article-title>SciPy 1.0: fundamental algorithms for scientific computing in Python</article-title>. <source>Nat Methods</source>. (<year>2020</year>) <volume>17</volume>:<page-range>261&#x2013;72</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41592&#x2013;019-0686&#x2013;2</pub-id>
</citation>
</ref>
<ref id="B60">
<label>60</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sundararajan</surname> <given-names>M</given-names>
</name>
<name>
<surname>Taly</surname> <given-names>A</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>Q</given-names>
</name>
</person-group>. <source>Axiomatic Attribution for Deep Networks</source>. (<year>2017</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1703.01365</pub-id>.</citation>
</ref>
<ref id="B61">
<label>61</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tareen</surname> <given-names>A</given-names>
</name>
<name>
<surname>Kinney</surname> <given-names>JB</given-names>
</name>
</person-group>. <article-title>Logomaker: beautiful sequence logos in Python</article-title>. <source>Bioinformatics</source>. (<year>2020</year>) <volume>36</volume>:<page-range>2272&#x2013;4</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btz921</pub-id>
</citation>
</ref>
<ref id="B62">
<label>62</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Korhonen</surname> <given-names>JH</given-names>
</name>
<name>
<surname>Palin</surname> <given-names>K</given-names>
</name>
<name>
<surname>Taipale</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ukkonen</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>Fast motif matching revisited: high-order PWMs, SNPs and indels</article-title>. <source>Bioinformatics</source>. (<year>2017</year>) <volume>33</volume>:<page-range>514&#x2013;21</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btw683</pub-id>
</citation>
</ref>
<ref id="B63">
<label>63</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Castro-Mondragon</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Riudavets-Puig</surname> <given-names>R</given-names>
</name>
<name>
<surname>Rauluseviciute</surname> <given-names>I</given-names>
</name>
<name>
<surname>Berhanu Lemma</surname> <given-names>R</given-names>
</name>
<name>
<surname>Turchi</surname> <given-names>L</given-names>
</name>
<name>
<surname>Blanc-Mathieu</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>JASPAR 2022: the 9th release of the open-access database of transcription factor binding profiles</article-title>. <source>Nucleic Acids Res</source>. (<year>2022</year>) <volume>50</volume>:<page-range>D165&#x2013;73</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/NAR/GKAB1113</pub-id>
</citation>
</ref>
<ref id="B64">
<label>64</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>D</given-names>
</name>
<name>
<surname>Dhiman</surname> <given-names>V</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>P</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>J</given-names>
</name>
<name>
<surname>McGillivray</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>An integrative ENCODE resource for cancer genomics</article-title>. <source>Nat Commun</source>. (<year>2020</year>) <volume>11</volume>:<fpage>3696</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-020-14743-w</pub-id>
</citation>
</ref>
<ref id="B65">
<label>65</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Langmead</surname> <given-names>B</given-names>
</name>
<name>
<surname>Salzberg</surname> <given-names>SL</given-names>
</name>
</person-group>. <article-title>Fast gapped-read alignment with Bowtie 2</article-title>. <source>Nat Methods 2012 9:4</source>. (<year>2012</year>) <volume>9</volume>:<page-range>357&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nmeth.1923</pub-id>
</citation>
</ref>
<ref id="B66">
<label>66</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Danecek</surname> <given-names>P</given-names>
</name>
<name>
<surname>Bonfield</surname> <given-names>JK</given-names>
</name>
<name>
<surname>Liddle</surname> <given-names>J</given-names>
</name>
<name>
<surname>Marshall</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ohan</surname> <given-names>V</given-names>
</name>
<name>
<surname>Pollard</surname> <given-names>MO</given-names>
</name>
<etal/>
</person-group>. <article-title>Twelve years of SAMtools and BCFtools</article-title>. <source>Gigascience</source>. (<year>2021</year>) <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/gigascience/giab008</pub-id>
</citation>
</ref>
<ref id="B67">
<label>67</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>T</given-names>
</name>
<name>
<surname>Meyer</surname> <given-names>CA</given-names>
</name>
<name>
<surname>Eeckhoute</surname> <given-names>J</given-names>
</name>
<name>
<surname>Johnson</surname> <given-names>DS</given-names>
</name>
<name>
<surname>Bernstein</surname> <given-names>BE</given-names>
</name>
<etal/>
</person-group>. <article-title>Model-based analysis of chIP-seq (MACS)</article-title>. <source>Genome Biol</source>. (<year>2008</year>) <volume>9</volume>:<fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/gb-2008-9-9-r137</pub-id>
</citation>
</ref>
<ref id="B68">
<label>68</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heinz</surname> <given-names>S</given-names>
</name>
<name>
<surname>Benner</surname> <given-names>C</given-names>
</name>
<name>
<surname>Spann</surname> <given-names>N</given-names>
</name>
<name>
<surname>Bertolino</surname> <given-names>E</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>YC</given-names>
</name>
<name>
<surname>Laslo</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Simple combinations of lineage-determining transcription factors prime cis-regulatory elements required for macrophage and B cell identities</article-title>. <source>Mol Cell</source>. (<year>2010</year>) <volume>38</volume>:<page-range>576&#x2013;89</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.molcel.2010.05.004</pub-id>
</citation>
</ref>
<ref id="B69">
<label>69</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ram&#xed;rez</surname> <given-names>F</given-names>
</name>
<name>
<surname>Ryan</surname> <given-names>DP</given-names>
</name>
<name>
<surname>Gr&#xfc;ning</surname> <given-names>B</given-names>
</name>
<name>
<surname>Bhardwaj</surname> <given-names>V</given-names>
</name>
<name>
<surname>Kilpert</surname> <given-names>F</given-names>
</name>
<name>
<surname>Richter</surname> <given-names>AS</given-names>
</name>
<etal/>
</person-group>. <article-title>deepTools2: a next generation web server for deep-sequencing data analysis</article-title>. <source>Nucleic Acids Res</source>. (<year>2016</year>) <volume>44</volume>:<page-range>W160&#x2013;5</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/NAR/GKW257</pub-id>
</citation>
</ref>
<ref id="B70">
<label>70</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thientosapol</surname> <given-names>ES</given-names>
</name>
<name>
<surname>Sharbeen</surname> <given-names>G</given-names>
</name>
<name>
<surname>Edwin Lau</surname> <given-names>KK</given-names>
</name>
<name>
<surname>Bosnjak</surname> <given-names>D</given-names>
</name>
<name>
<surname>Durack</surname> <given-names>T</given-names>
</name>
<name>
<surname>Stevanovski</surname> <given-names>I</given-names>
</name>
<etal/>
</person-group>. <article-title>Proximity to AGCT sequences dictates MMR-independent versus MMR-dependent mechanisms for AID-induced mutation via UNG2</article-title>. <source>Nucleic Acids Res</source>. (<year>2017</year>) <volume>45</volume>:<page-range>3146&#x2013;57</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/NAR/GKW1300</pub-id>
</citation>
</ref>
<ref id="B71">
<label>71</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sale</surname> <given-names>JE</given-names>
</name>
<name>
<surname>Neuberger</surname> <given-names>MS</given-names>
</name>
</person-group>. <article-title>TdT-accessible breaks are scattered over the immunoglobulin V domain in a constitutively hypermutating B cell line</article-title>. <source>Immunity</source>. (<year>1998</year>) <volume>9</volume>:<page-range>859&#x2013;69</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S1074&#x2013;7613(00)80651&#x2013;2</pub-id>
</citation>
</ref>
<ref id="B72">
<label>72</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rodriguez</surname> <given-names>OL</given-names>
</name>
<name>
<surname>Gibson</surname> <given-names>WS</given-names>
</name>
<name>
<surname>Parks</surname> <given-names>T</given-names>
</name>
<name>
<surname>Emery</surname> <given-names>M</given-names>
</name>
<name>
<surname>Powell</surname> <given-names>J</given-names>
</name>
<name>
<surname>Strahl</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>A novel framework for characterizing genomic haplotype diversity in the human immunoglobulin heavy chain locus</article-title>. <source>Front Immunol</source>. (<year>2020</year>) <volume>11</volume>:<elocation-id>2136</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2020.02136</pub-id>
</citation>
</ref>
<ref id="B73">
<label>73</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>M</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>P</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>L</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>J</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Profiling the peripheral blood T cell receptor repertoires of gastric cancer patients</article-title>. <source>Front Immunol</source>. (<year>2022</year>) <volume>13</volume>:<elocation-id>848113</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2022.848113</pub-id>
</citation>
</ref>
<ref id="B74">
<label>74</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grundstr&#xf6;m</surname> <given-names>C</given-names>
</name>
<name>
<surname>Grundstr&#xf6;m</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>The transcription factor E2A can bind to and cleave single-stranded immunoglobulin heavy chain locus DNA</article-title>. <source>Mol Immunol</source>. (<year>2023</year>) <volume>153</volume>:<page-range>51&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.molimm.2022.11.013</pub-id>
</citation>
</ref>
<ref id="B75">
<label>75</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parsa</surname> <given-names>JY</given-names>
</name>
<name>
<surname>Ramachandran</surname> <given-names>S</given-names>
</name>
<name>
<surname>Zaheen</surname> <given-names>A</given-names>
</name>
<name>
<surname>Nepal</surname> <given-names>RM</given-names>
</name>
<name>
<surname>Kapelnikov</surname> <given-names>A</given-names>
</name>
<name>
<surname>Belcheva</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Negative supercoiling creates single-stranded patches of DNA that are substrates for AID&#x2013;mediated mutagenesis</article-title>. <source>PloS Genet</source>. (<year>2012</year>) <volume>8</volume>:<elocation-id>e1002518</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/JOURNAL.PGEN.1002518</pub-id>
</citation>
</ref>
<ref id="B76">
<label>76</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zan</surname> <given-names>H</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Komori</surname> <given-names>A</given-names>
</name>
<name>
<surname>Holloman</surname> <given-names>WK</given-names>
</name>
<name>
<surname>Casali</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>AID-Dependent generation of resected double-strand DNA breaks and recruitment of Rad52/Rad51 in Somatic hypermutation</article-title>. <source>Immunity</source>. (<year>2003</year>) <volume>18</volume>:<page-range>727&#x2013;38</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S1074&#x2013;7613(03)00151&#x2013;1</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>