<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2017.00537</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Gene-Specific Substitution Profiles Describe the Types and Frequencies of Amino Acid Changes during Antibody Somatic Hypermutation</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Sheng</surname> <given-names>Zizhang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="cor1">&#x0002A;</xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/371221"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Schramm</surname> <given-names>Chaim A.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/375370"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Kong</surname> <given-names>Rui</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author" id="collab1">
<collab>NISC Comparative Sequencing Program</collab>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Mullikin</surname> <given-names>James C.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Mascola</surname> <given-names>John R.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Kwong</surname> <given-names>Peter D.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Shapiro</surname> <given-names>Lawrence</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="cor1">&#x0002A;</xref>
<uri xlink:href="http://frontiersin.org/people/u/354152"/>
</contrib>
</contrib-group>
<contrib-group content-type="collab-list">
<contrib contrib-type="collab" rid="collab1">
<name><surname>Benjamin</surname> <given-names>Betty</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Bouffard</surname> <given-names>Gerry</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Brooks</surname> <given-names>Shelise</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Coleman</surname> <given-names>Holly</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Dekhtyar</surname> <given-names>Mila</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Guan</surname> <given-names>Xiaobin</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Han</surname> <given-names>Joel</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Ho</surname> <given-names>Shi- ling</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Legaspi</surname> <given-names>Richelle</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Maduro</surname> <given-names>Quino</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Masiello</surname> <given-names>Cathy</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>McDowell</surname> <given-names>Jenny</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Montemayor</surname> <given-names>Casandra</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Mullikin</surname> <given-names>James</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Park</surname> <given-names>Morgan</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Riebow</surname> <given-names>Nancy</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Rosarda</surname> <given-names>Jessica</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Schandler</surname> <given-names>Karen</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Schmidt</surname> <given-names>Brian</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Sison</surname> <given-names>Christina</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Smith</surname> <given-names>Ray</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Stantripop</surname> <given-names>Mal</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Thomas</surname> <given-names>James</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Thomas</surname> <given-names>Pam</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Vemulapalli</surname> <given-names>Meg</given-names></name>
</contrib>
<contrib contrib-type="collab" rid="collab1">
<name><surname>Young</surname> <given-names>Alice</given-names></name>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Biochemistry and Molecular Biophysics, Columbia University</institution>, <addr-line>New York, NY</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Systems Biology, Columbia University</institution>, <addr-line>New York, NY</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Vaccine Research Center, National Institute of Allergy and Infectious Diseases, National Institutes of Health</institution>, <addr-line>Bethesda, MD</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>NIH Intramural Sequencing Center, National Human Genome Research Institute, National Institutes of Health</institution>, <addr-line>Bethesda, MD</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Deborah K. Dunn-Walters, University of Surrey, UK</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Patricia Johanna Gearhart, National Institutes of Health, United States; Frederick Matsen, Fred Hutchinson Cancer Research Center, United States; Almudena R. Ramiro, Spanish National Centre for Cardiovascular Research, Spain</p></fn>
<corresp content-type="corresp" id="cor1">&#x0002A;Correspondence: Zizhang Sheng, <email>zs2248&#x00040;c2b2.columbia.edu</email>; Lawrence Shapiro, <email>shapiro&#x00040;convex.hhmi.columbia.edu</email></corresp>
<fn fn-type="other" id="fn001"><p><sup>&#x02020;</sup>These authors have contributed equally to this work.</p></fn>
<fn fn-type="other" id="fn002"><p>Specialty section: This article was submitted to B Cell Biology, a section of the journal Frontiers in Immunology</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>10</day>
<month>05</month>
<year>2017</year>
</pub-date>
<pub-date pub-type="collection">
<year>2017</year>
</pub-date>
<volume>8</volume>
<elocation-id>537</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>02</month>
<year>2017</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>04</month>
<year>2017</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2017 Sheng, Schramm, Kong, NISC Comparative Sequencing Program, Mullikin, Mascola, Kwong and Shapiro.</copyright-statement>
<copyright-year>2017</copyright-year>
<copyright-holder>Sheng, Schramm, Kong, NISC Comparative Sequencing Program, Mullikin, Mascola, Kwong and Shapiro</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) or licensor are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Somatic hypermutation (SHM) plays a critical role in the maturation of antibodies, optimizing recognition initiated by recombination of V(D)J genes. Previous studies have shown that the propensity to mutate is modulated by the context of surrounding nucleotides and that SHM machinery generates biased substitutions. To investigate the intrinsic mutation frequency and substitution bias of SHMs at the amino acid level, we analyzed functional human antibody repertoires and developed mGSSP (method for gene-specific substitution profile), a method to construct amino acid substitution profiles from next-generation sequencing-determined B cell transcripts. We demonstrated that these gene-specific substitution profiles (GSSPs) are unique to each V gene and highly consistent between donors. We also showed that the GSSPs constructed from functional antibody repertoires are highly similar to those constructed from antibody sequences amplified from non-productively rearranged passenger alleles, which do not undergo functional selection. This suggests the types and frequencies, or mutational space, of a majority of amino acid changes sampled by the SHM machinery to be well captured by GSSPs. We further observed the rates of mutational exchange between some amino acids to be both asymmetric and context dependent and to correlate weakly with their biochemical properties. GSSPs provide an improved, position-dependent alternative to standard substitution matrices, and can be utilized to developing software for accurately modeling the SHM process. GSSPs can also be used for predicting the amino acid mutational space available for antigen-driven selection and for understanding factors modulating the maturation pathways of antibody lineages in a gene-specific context. The mGSSP method can be used to build, compare, and plot GSSPs<xref ref-type="fn" rid="fn1"><sup>1</sup></xref>; we report the GSSPs constructed for 69 common human V genes (DOI: <uri xlink:href="https://doi.org/10.6084/m9.figshare.3511083">10.6084/m9.figshare.3511083</uri>) and provide high-resolution logo plots for each (DOI: <uri xlink:href="https://doi.org/10.6084/m9.figshare.3511085">10.6084/m9.figshare.3511085</uri>).</p>
</abstract>
<kwd-group>
<kwd>antibodyomics</kwd>
<kwd>B cell ontogeny</kwd>
<kwd>broadly neutralizing antibody</kwd>
<kwd>mutation frequency</kwd>
<kwd>repertoire diversity</kwd>
</kwd-group>
<contract-num rid="cn01">AI104722-3, R01 AI104387-3, S10OD012351, S10OD021764</contract-num>
<contract-sponsor id="cn01">National Institutes of Health<named-content content-type="fundref-id">10.13039/100000002</named-content></contract-sponsor>
<counts>
<fig-count count="6"/>
<table-count count="0"/>
<equation-count count="4"/>
<ref-count count="48"/>
<page-count count="14"/>
<word-count count="10893"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="introduction">
<title>Introduction</title>
<p>The variable regions of B cell receptors are responsible for antigen recognition and are generated by V(D)J recombination in the bone marrow. This generates a diverse initial pool of B cell receptors, but their affinities for antigens are usually low. Thus, somatic hypermutation (SHM) in the immunoglobulin (Ig) variable region is a key process for increasing affinity. SHM mainly occurs during B cell proliferation at the germinal center and is predominantly initiated by activation-induced cytidine deaminase (AID) (<xref ref-type="bibr" rid="B1">1</xref>&#x02013;<xref ref-type="bibr" rid="B5">5</xref>), which deaminates cytosine to uracil. The resulting U&#x02022;G mismatch undergoes error-prone or error-free repair by DNA repair pathways or by DNA replication (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B6">6</xref>&#x02013;<xref ref-type="bibr" rid="B9">9</xref>), and a mutation can be introduced at the targeted U&#x02022;G pair or a downstream A&#x02022;T pair (<xref ref-type="bibr" rid="B6">6</xref>, <xref ref-type="bibr" rid="B10">10</xref>). AID preferentially acts at hotspot motifs such as WR<underline>C</underline>Y (W&#x02009;&#x0003D;&#x02009;A or T, R&#x02009;&#x0003D;&#x02009;A or G, Y&#x02009;&#x0003D;&#x02009;C or T) (<xref ref-type="bibr" rid="B5">5</xref>, <xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>), while avoiding coldspot motifs such as SY<underline>C</underline> (S&#x02009;&#x0003D;&#x02009;C or G) (<xref ref-type="bibr" rid="B13">13</xref>). Thus, the distribution of hotspot and coldspot motifs, or the intrinsic mutability of a germline gene, modulates where mutations occur. Moreover, the targeted nucleotide positions are more likely to result in transition mutations (A&#x02194;G, C&#x02194;T) than transversion mutations (A&#x02194;C, A&#x02194;T, C&#x02194;G, G&#x02194;T) (<xref ref-type="bibr" rid="B8">8</xref>), further indicating that the types and frequencies of mutations (mutational space) are not sampled randomly at the nucleotide level.</p>
<p>Several previous studies have attempted to capture these biases with models based on 2-, 3-, 4-, 5-, and 7-nucleotide motifs (<xref ref-type="bibr" rid="B14">14</xref>&#x02013;<xref ref-type="bibr" rid="B19">19</xref>) or based on observed amino acid substitutions (<xref ref-type="bibr" rid="B20">20</xref>). Such substitution models provide important means for annotating antibody sequences (<xref ref-type="bibr" rid="B21">21</xref>), calculating genetic diversity (<xref ref-type="bibr" rid="B20">20</xref>) and evaluating selection pressure (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B23">23</xref>). Nonetheless, recent studies have revealed that the mutability of a nucleotide motif can vary between complementarity determining regions (CDRs) and framework regions (FWRs) (<xref ref-type="bibr" rid="B12">12</xref>) and that mutability and mutation biases are position, chain, and species dependent (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B24">24</xref>). Thus, models considering all context-dependent factors are required to characterize the SHM process precisely.</p>
<p>Patterns of amino acid mutations resulting from SHM biases at the nucleotide level are of great interest, as they determine antibody functionality. However, the subset of possible amino acid changes that are explored during affinity maturation has not been systematically investigated. This is partially because variations in the mutability and substitution bias of the three nucleotide positions of a codon lead to complications in the prediction of mutability and substitution biases of an amino acid position. In addition, the effects of functional selection are difficult to predict <italic>a priori</italic>, especially when the cognate antigen is not already known. Nonetheless, in a previous study, we observed that antibodies originating from the same germline V gene shared &#x0007E;20% of their amino acid substitutions in the V region, irrespective of antigen specificity (<xref ref-type="bibr" rid="B25">25</xref>). Similarly, in a recent vaccination study, we showed that the high-frequency substitutions observed in the V region of IGHV1-2-derived anti-Env antibodies also appear with high frequency in IGHV1-2-derived antibodies that do not target Env (<xref ref-type="bibr" rid="B26">26</xref>). Together, these results suggest that the occurrence frequencies of amino acid substitutions in antibody repertoires are at least partially independent of antigen-driven selection, and that the intrinsic mutability and substitution bias of a germline gene is a dominant factor modulating SHM. This suggests the possibility of using gene-specific substitution profiles (GSSPs), which implicitly incorporate all context-dependent factors regulating SHM, to predict the mutational space sampled by SHM machinery. Such investigation is important for understanding substitution patterns observed in antigen-specific antibodies and predicting the chance of re-eliciting similar SHM patterns by vaccination.</p>
<p>In this study, we describe a new software tool for examining the sampled amino acid mutational space for each position of germline V genes. We demonstrate the usefulness of this technique by analyzing the antibody repertoires of six human donors and demonstrating that the GSSPs constructed using substitutions from functional antibodies and recapitulate the mutational space sampled by SHM machinery. We show that the mutational space of a V gene is not sampled uniformly and that the sampling bias is gene specific and similar among donors and over time. The software for constructing and analyzing GSSPs is available from GitHub as part of the SONAR suite (<xref ref-type="bibr" rid="B27">27</xref>).</p>
</sec>
<sec id="S2">
<title>Results</title>
<sec id="S2-1">
<title>Construction of Robust GSSPs for V Genes</title>
<p>To investigate the mutational space sampled by each germline V gene, we first compared the somatic mutations observed in human antibody repertoires of three healthy donors and three HIV-1-infected donors. Briefly, B cell receptor transcripts from peripheral blood B cells were sequenced using either Roche 454 pyrosequencing or Illumina MiSeq technologies. Starting from hundreds of thousands or millions of reads from each sample (Table S1), we assigned germline V and J genes for each transcript, removed low-quality sequences, identified clonal lineages, and selected one representative sequence per clonal lineage to build GSSPs for each V gene (Figures <xref ref-type="fig" rid="F1">1</xref> and <xref ref-type="fig" rid="F2">2</xref>). We used the program partis (<xref ref-type="bibr" rid="B21">21</xref>) to predict novel germline alleles for all quality-filtered repertoires (see <xref ref-type="sec" rid="S4">Materials and Methods</xref>), providing high confidence that all germline gene polymorphisms were excluded from the profiles.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p><bold>Flowchart for the analyses of human antibody repertoires and construction of gene-specific substitution profile (GSSP)</bold>. The next-generation sequencing datasets were first processed using SONAR to filter out low-quality reads and to assign V(D)J genes for each transcript. Novel alleles were identified using partis. SONAR was then used to identify antibody clones, and one representative sequence was chosen in each clone for building GSSPs using method for gene-specific substitution profile.</p></caption>
<graphic xlink:href="fimmu-08-00537-g001.tif"/>
</fig>
<p>In order to test the robustness of GSSPs to noise in the data, we subsampled the lineages found in each of the three healthy donors and built profiles for common VH genes using 25, 50, 100, 200, or 300 lineages per profile. We then calculated the Jensen&#x02013;Shannon divergence between GSSPs (see <xref ref-type="sec" rid="S4">Materials and Methods</xref>). We found that the between-donor Jensen&#x02013;Shannon divergence between these profiles began to converge when 300 lineages were used to create a profile (Figure <xref ref-type="supplementary-material" rid="SM1">S1</xref>A in Supplementary Material). We therefore used 300 as the minimum number of lineages to build a mutational profile for further analyses except for quantifying the similarities of GSSPs (see next section), in which seven of 69 GSSPs built using 100&#x02013;300 lineages were included but they did not change our conclusions. We also determined that there were no significant differences between GSSPs from IgM and IgG repertories, and therefore treated all VH data together. High-resolution plots of each profile, as well as the underlying numerical data, can be found at DOIs 10.6084/m9.figshare.3511083 and 10.6084/m9.figshare.3511085.</p>
<p>For each position of each germline V gene (numbered using IMGT scheme), we calculated the rarity of all possible substitutions as
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>V</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>*</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>V</mml:mi><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x0200B;</mml:mtext><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:math></disp-formula>
where <italic>m<sub>V,i</sub></italic> is the substitution frequency of position <italic>i</italic> in germline gene <italic>V</italic> and <italic>f<sub>Vi,a</sub></italic> is the substitution bias, or frequency at which a particular non-germline amino acid <italic>a</italic> was observed in all <italic>V</italic> gene lineages with substitutions at position <italic>i</italic>. The rarity is undefined for germline amino acids and equal to 1.0 for substitutions that are never observed in a particular dataset. When we examined the effects of choosing different representative sequences for each lineage (see <xref ref-type="sec" rid="S4">Materials and Methods</xref>), the rarity scores of all substitutions were highly correlated regardless of representative sequence (Figure <xref ref-type="supplementary-material" rid="SM1">S1</xref>B in Supplementary Material), demonstrating again that the observed GSSPs are not significantly affected by noise in the data.</p>
</sec>
<sec id="S2-2">
<title>Substitution Profile of Each Germline V Gene Is Similar among Donors and Over Time</title>
<p>Figure <xref ref-type="fig" rid="F2">2</xref> shows the GSSPs of IGHV1-69 and IGKV3-20 from three healthy donors in which the distributions of somatic mutation levels of lineages for each gene are similar (Figure <xref ref-type="supplementary-material" rid="SM2">S2</xref>A in Supplementary Material). Overall, the amino acid GSSPs of each V gene are remarkably consistent, similar to previous studies of mutations and selection (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B18">18</xref>, <xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B21">21</xref>) In contrast, different V genes have noticeably divergent profiles, even within a single donor, as can be seen from the GSSPs of IGHV1-2 and IGHV1-69 (Figures <xref ref-type="supplementary-material" rid="SM2">S2</xref>B,C in Supplementary Material). The GSSPs depend on two factors, substitution frequency and substitution bias, each of which is dealt with separately in the following sections. We also combine these two to define rarity, shown in Equation <xref ref-type="disp-formula" rid="E1">1</xref>.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p><bold>Inter-donor similarities of gene-specific substitution profiles (GSSPs) of IGHV1-69 and IGKV3-20 germline genes</bold>. At each position of the GSSPs, the length of an amino acid letter represents the frequency of substitution observed in a repertoire, with the germline amino acids showed at the bottom of each panel. For each gene, the GSSPs [<bold>(A)</bold> for IGHV1-69 and <bold>(B)</bold> for IGKV3-20] are similar among three healthy donors. For each position of each donor, the substitutions were colored by the physicochemical properties of the amino acids. Blue: Arg, Lys, His; red, Asp, Glu; green, Gly, Ser, Thr, Tyr, Cys; black, Pro, Ala, Trp, Phe, Leu, Ile, Met, Val; purple, Asn, Gln. See also Figures <xref ref-type="supplementary-material" rid="SM1">S1</xref> and <xref ref-type="supplementary-material" rid="SM2">S2</xref> in Supplementary Material.</p></caption>
<graphic xlink:href="fimmu-08-00537-g002.tif"/>
</fig>
<p>The substitution frequency at each position of each V gene is similar among donors. As expected, the substitution frequency is higher within CDR 1 and CDR2 (Figure <xref ref-type="fig" rid="F2">2</xref>, as defined by IMGT), consistent with the fact that the CDRs contain higher proportions of AID hotspot motifs (<xref ref-type="bibr" rid="B18">18</xref>). For IGHV1-69, six positions in framework region (FWR) 3 exhibited substitution frequencies as high as positions in the CDRs (&#x02265;30%), probably because substitutions in the FWRs can play important roles in recognizing antigens and regulating antibody structural stability and conformations (<xref ref-type="bibr" rid="B28">28</xref>&#x02013;<xref ref-type="bibr" rid="B31">31</xref>). Moreover, we also observed high frequencies of substitutions at positions adjacent to the CDRs (position 39 of IGHV1-69 and position 66 of IGKV3-20, IMGT numbering). Since these positions connect the CDRs to the &#x003B2;-strands of the FWRs, it is possible that these positions are important for regulating the flexibility and conformations of the CDR loops (<xref ref-type="bibr" rid="B28">28</xref>, <xref ref-type="bibr" rid="B29">29</xref>).</p>
<p>For many positions, we observed 1&#x02013;3 dominant substitutions with much higher frequency than other substitutions, indicating a substantial substitution bias. As expected, a major bias is toward substitutions that require only a single nucleotide change (Figure <xref ref-type="fig" rid="F3">3</xref>; Figures <xref ref-type="supplementary-material" rid="SM3">S3</xref>A,B in Supplementary Material). In addition, we found that amino acid substitutions requiring 2 or 3 nucleotide changes are significantly rarer than those which require only a single nucleotide change (Figure <xref ref-type="supplementary-material" rid="SM3">S3</xref>C in Supplementary Material). However, the exact substitutions that are preferred vary based on context, even for a specific codon (Figure <xref ref-type="fig" rid="F3">3</xref>). This is consistent with previous findings that mutation frequency varies by position within the V gene, independent of and in addition to variations due to SHM hotspots and coldspots (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B17">17</xref>).</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p><bold>Substitution frequency and substitution bias of codon TCC in different VH genes</bold>. The gene-specific substitution profiles (GSSPs) of all TCC codons (encoding Serine) in our VH germline database. Codons were first sorted by VH family. Within each VH family, the GSSPs of homologous positions of different VH genes are shown together while those of non-homologous positions are separated by a space. The comparison showed that the GSSPs of TCC codons are more similar at homologous positions than between non-homologous positions, suggesting the substitution frequency and substitution bias of TCC codon are dependent on the nucleotide context. A similar nucleotide context dependency was also observed for other codons (Figures <xref ref-type="supplementary-material" rid="SM3">S3</xref>A,B in Supplementary Material). Color scheme: green, amino acid replacement involves single nucleotide mutation; yellow, amino acid replacement involves two nucleotide mutations; red, amino acid replacement involves three nucleotide mutations. See also Figure <xref ref-type="supplementary-material" rid="SM3">S3</xref> in Supplementary Material.</p></caption>
<graphic xlink:href="fimmu-08-00537-g003.tif"/>
</fig>
<p>To quantitatively compare different GSSPs, we calculated weighted average of the Jensen&#x02013;Shannon divergence between homologous positions over the entire V gene (see <xref ref-type="sec" rid="S4">Materials and Methods</xref>). We then used multidimensional scaling (MDS) to visualize these distances for a subset of most frequently used V<sub>H</sub> (Figure <xref ref-type="fig" rid="F4">4</xref>A) and V<sub>&#x003BA;</sub> genes (Figure <xref ref-type="fig" rid="F4">4</xref>B). (MDS plots of all V<sub>H</sub>, V<sub>&#x003BA;</sub>, and V<sub>&#x003BB;</sub> genes are shown in Figures <xref ref-type="supplementary-material" rid="SM4">S4</xref>A&#x02013;C in Supplementary Material). These plots confirmed that the GSSPs of each V gene from all donors clustered together and that the GSSPs of each V gene family are more similar than between V gene families. Moreover, we did not observe any differences based on HIV status or the sequencing technology used to obtain data. The GSSPs of a V gene are also similar across longitudinal samples of the same donor (Figures <xref ref-type="fig" rid="F4">4</xref>C,D; Figures <xref ref-type="supplementary-material" rid="SM4">S4</xref>D,E in Supplementary Material). Finally, the distributions of substitution frequency and rarity were both highly correlated between donors (Pearson&#x02019;s <italic>r</italic> &#x0007E;0.9 and &#x0007E;0.8, respectively) (Figures <xref ref-type="fig" rid="F4">4</xref>E,F; Figures <xref ref-type="supplementary-material" rid="SM5">S5</xref>A,B in Supplementary Material). In order to increase our sampling depth, we therefore combined all lineages from the three healthy donors and generated a single set of GSSPs. We recalculated substitution frequency and rarity from these profiles, which were used for all further analyses.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p><bold>Similarity of gene-specific substitution profiles (GSSPs) between donors and across time</bold>. The Jensen&#x02013;Shannon divergence was used to compare different GSSPs, and the resulting matrix of distances was visualized using multidimensional scaling. This showed that profiles from the same germline genes in different donors are more similar than profiles from closely related germline genes in the same donor for both <bold>(A)</bold> heavy chain and <bold>(B)</bold> kappa chain V genes. Similar analyses for lambda chain V genes are shown in Figure <xref ref-type="supplementary-material" rid="SM4">S4</xref> in Supplementary Material. Note that there is no discernable difference between profiles derived from HIV<sup>&#x02212;</sup> donors 08248, 23810, and 32647 and those derived from HIV<sup>&#x0002B;</sup> donors CAP256, NIH45, and Z258. Furthermore, data from donor Z258 (tan symbols) were obtained using Illumina MiSeq sequencing technology, while all other datasets were collected using Roche 454 pyrosequencing, but no difference is observed based on platform. In addition to similarity between donors, the profile from each germline gene is consistent over time, as shown for donor CAP256 <bold>(C)</bold> heavy chains and <bold>(D)</bold> lambda chains. Across all heavy chain V genes, <bold>(E)</bold> positional substitution frequency and <bold>(F)</bold> the rarity of each possible substitution are strongly correlated between donors (see also Figures <xref ref-type="supplementary-material" rid="SM4">S4</xref> and <xref ref-type="supplementary-material" rid="SM5">S5</xref> in Supplementary Material).</p></caption>
<graphic xlink:href="fimmu-08-00537-g004.tif"/>
</fig>
</sec>
<sec id="S2-3">
<title>Major Factors in Determining Position-Specific Substitution Preferences Observed in GSSP</title>
<p>The fact that similar preferred substitutions are detected in the repertoires of multiple donors could be explained by a limited range of possibilities. One possibility, though unlikely, is that observed GSSPs are dominated by convergent selection against common antigens. It is also possible that antigen-driven selection, while critical for the development of each individual lineage, is &#x0201C;averaged out&#x0201D; over the entire repertoire such that the observed GSSPs reflect the biased action of the underlying SHM machinery. Finally, it is possible that constraints on structural stability and other sources could limit the substitutional space explored by the antibody repertoire.</p>
<p>In order to distinguish between these possibilities, we first built a GSSP using lineages derived from non-productive rearrangements of IGHV3-23 (<xref ref-type="bibr" rid="B32">32</xref>). Because these sequences are derived from the &#x0201C;passenger&#x0201D; allele, they are not subject to selective pressure, even though AID continues to act on them (<xref ref-type="bibr" rid="B6">6</xref>, <xref ref-type="bibr" rid="B33">33</xref>). We find that the GSSPs of the functional and non-productive repertoires are highly similar (Figure <xref ref-type="fig" rid="F5">5</xref>A), although mutations to cysteine observed at several CDR2 and FWR3 positions were suppressed in the GSSP of functional antibodies. We next compared the substitution frequency at each position of the two profiles and showed that they are as similar to each other as between functional antibody profiles of two donors (Pearson&#x02019;s <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.90, <italic>p</italic>&#x02009;&#x0003C;&#x0003C;&#x02009;0.01) (Figures <xref ref-type="fig" rid="F5">5</xref>B and <xref ref-type="fig" rid="F4">4</xref>E). The same was true for rarity (Pearson&#x02019;s <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.80, <italic>p</italic>&#x02009;&#x0003C;&#x0003C;&#x02009;0.01) (Figures <xref ref-type="fig" rid="F5">5</xref>C and <xref ref-type="fig" rid="F4">4</xref>F). While the residual differences between the functional and non-productive GSSPs are likely to reflect the effect of antigen-driven selection, the strong overall correlation suggests that antigen-driven selective effects are mostly averaged out when calculating GSSPs across the entire functional repertoire. Indeed, calculations of rarity based on two single lineages showed much lower correlations to their respective functional repertoires (Pearson&#x02019;s <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.11, <italic>p</italic>&#x02009;&#x0003D;&#x02009;0.11 for lineage 08248-00037 and <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.26, <italic>p</italic>&#x02009;&#x0003D;&#x02009;1e-6 for lineage CAP256-VRC26) (Figure <xref ref-type="supplementary-material" rid="SM6">S6</xref> in Supplementary Material), demonstrating that we can observe the modulation of substitution preferences for individual lineages by antigen-driven selection.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p><bold>The action of functional selection on individual lineages is averaged out when determining position-specific mutation preferences in antibody repertoires</bold>. <bold>(A)</bold> Logo plots showing the IGHV3-23 gene-specific substitution profile (GSSP) for function antibody repertoires (top), non-productively rearranged passenger alleles (middle), and a simulation of SHM using the SF5 model (bottom). Despite the absence of selective pressure on the lineages derived from passenger alleles, the profile is quite similar to that of the functional repertoire. In addition, the simulated repertoire also successfully recapitulates the dominant mutations. This suggests that the action of selection on lineages in the functional repertoire is averaged out when constructing the GSSP from an entire repertoire. <bold>(B)</bold> Pairwise comparisons of substitution frequency among functional antibody repertoires, passenger alleles, and simulated lineages under the S5F model. The comparisons showed that the mutation frequency of an amino acid position is modulated dominantly by SHM machinery. <bold>(C)</bold> Pairwise comparisons of rarity among functional antibody repertoires, passenger alleles, and simulated lineages under the S5F model. The comparisons showed that the substitution bias observed at an amino acid position is modulated dominantly by SHM machinery. See also Figure <xref ref-type="supplementary-material" rid="SM6">S6</xref> in Supplementary Material.</p></caption>
<graphic xlink:href="fimmu-08-00537-g005.tif"/>
</fig>
<p>Previous work has found that, when site-specific selection coefficients for functional sequences are compared to those derived from non-productive sequences, approximately 30 and 5% of positions in the framework 3 region of human heavy chain genes are under negative and positive selection, respectively (<xref ref-type="bibr" rid="B34">34</xref>), which is roughly consistent between donors. McCoy et al further demonstrated that positions under negative selection tend to have less exposed surface area and proposed that the types and frequencies of substitutions at these sites may be restrained by negative selection for structural stability. To understand whether selection for structural stability modulates the similarities of substitution preferences observed in GSSPs, we compared substitution frequencies and rarity scores in the framework 3 region of IGHV3-23 from functional and non-productive repertoires at positions inferred as being under positive, negative, or neutral selection by McCoy et al&#x02009;(Figure <xref ref-type="supplementary-material" rid="SM7">S7</xref> in Supplementary Material). The analysis revealed that the frequency of substitutions is less correlated between the functional and non-productive repertoires for positions under either positive or negative selection (Pearson&#x02019;s <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.70, 0.63, 0.30 for neutral, negative, and positive selection respectively), suggesting that selection modulates the substitution frequencies observed in GSSPs. The analysis further showed that sites under positive selection showed reduced but still significant correlation between the functional and non-productive repertoires (Pearson&#x02019;s <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.49 vs. <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.67 for neutral selection), suggesting selection modulates substitution bias in GSSPs. However, we were surprised to observe an increased correlation for sites under negative selection (Pearson&#x02019;s <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.78), but the correlation is comparable to the measured similarity of the rarity scores for all V gene positions between the functional and non-productive repertoires (Pearson&#x02019;s <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.79, Figure <xref ref-type="fig" rid="F5">5</xref>C). We note that only &#x0007E;38 sites with estimates of selection pressure by McCoy et al are also present in the set of non-productive sequences used in this study, which may allow the introduction of sampling bias. Nonetheless, the sites under neutral selection showed a high correlation coefficient between the functional and non-productive repertoires, suggesting that there are other factors modulating substitution preferences.</p>
<p>We next sought to determine whether the substitution frequency and substitution biases observed in GSSPs are modulated by the biased action of SHM machinery. To accomplish this, we conducted simulations and generated a virtual repertoire of IGHV3-23 lineages using S5F (<xref ref-type="bibr" rid="B18">18</xref>), an antibody-specific mutation model that estimates a mutability and mutation preference for each nucleotide based on the four surrounding nucleotides (two on either side). The comparisons of the GSSPs showed that the simulated repertoire reproduces many dominant mutations in the functional and passenger allele repertoires, such as G10D and V89I/L (Figure <xref ref-type="fig" rid="F5">5</xref>A). Indeed, the mutability and rarity of the simulated repertoire was reasonably similar to substitution frequency and rarity of the functional and passenger allele repertoire, respectively (Figures <xref ref-type="fig" rid="F5">5</xref>B,C). Conversely, simulations using a previous antibody-specific amino acid substitution model (AB) (<xref ref-type="bibr" rid="B20">20</xref>) showed a heavy bias toward the use of histidine. The Pearson&#x02019;s rho of the rarity scores derived from AB compared to those derived from the functional repertoires was very low (Figure <xref ref-type="supplementary-material" rid="SM8">S8</xref> in Supplementary Material). This is likely due to a number of causes, including the omission of positional effects (<xref ref-type="bibr" rid="B17">17</xref>) and the mixture of sequences from different loci and species used to build the model.</p>
<p>Since the S5F model was designed to reflect intrinsic AID hotspots and coldspots (<xref ref-type="bibr" rid="B18">18</xref>), the strong correlation of rarity scores between simulated and actual data provides additional evidence demonstrating that the molecular machinery of SHM plays a key role in producing the observed GSSPs. Because the substitutions observed in the GSSPs recapitulate the majority of the sampled mutational space, we will use the GSSPs to understand the features of mutations available for antigen-driven selection in analyses in the following sections.</p>
</sec>
<sec id="S2-4">
<title>Most Observed Mutations Are Rare but Can Be Sampled by High-SHM Lineages</title>
<p>We next investigated the distribution of the observed rarity of all possible mutations. Strikingly, over 70% of all possible mutations were never observed in any of our datasets. Moreover, nearly 45% of observed mutations were &#x0201C;extremely rare&#x0201D; (defined as rarity&#x02009;&#x0003E;&#x02009;0.995) (Figure <xref ref-type="fig" rid="F6">6</xref>A). Importantly, rarity is defined as a per-lineage score, as only one sequence per lineage is used to construct the GSSPs (Figure <xref ref-type="supplementary-material" rid="SM1">S1</xref>B in Supplementary Material). Thus, an extremely rare substitution is one that is expected to be sampled only by one in two 100 lineages. However, particular rare mutations can become fixed in highly expanded lineages (Figure <xref ref-type="supplementary-material" rid="SM6">S6</xref> in Supplementary Material) and therefore be present in a larger fraction of individual B cells.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p><bold>Most somatic hypermutations in the antibody repertoire are rare</bold>. Due to substitution biases, most observed mutations are seen only rarely. <bold>(A)</bold> The distribution of rarity scores for all observed mutations in all heavy chain V genes. For clarity, the top two bins are expanded in the inset. Approximately 85% of all mutations have a rarity of greater than 0.975 (rightmost bar in main plot); over 45% are extremely rare, with rarity greater than 0.995 (rightmost bar in inset). <bold>(B)</bold> The average substitution bias for each type of substitution shows a preference for conservative mutations and an asymmetric preference for certain mutations. <bold>(C)</bold> Substitution biases at individual positions are partially correlated with physicochemical properties (as measured by the BC0030 interchange matrix), but mutations with the same BC0030 score can have a wide range of substitution biases depending on genetic context. See also Figure <xref ref-type="supplementary-material" rid="SM7">S7</xref> in Supplementary Material.</p></caption>
<graphic xlink:href="fimmu-08-00537-g006.tif"/>
</fig>
<p>We then analyzed the substitution bias <italic>f<sub>Vi,a</sub></italic> for each substitution type at positions with substitution frequency <italic>m<sub>V,i</sub></italic>&#x02009;&#x02265;&#x02009;0.05 averaged over all germline genes (Figure <xref ref-type="fig" rid="F6">6</xref>B). We used the within-position bias rather than rarity to control for differences in overall mutabilities. The results demonstrated that conservative substitutions tend to be favored, but we observed many non-conservative substitutions such as R to T and P to S. Interestingly, we also observed many substitutions that were asymmetric. For instance, R to T is more common than T to R; E to D is more common than D to E. To quantify the difference between conservative and non-conservative mutations, we compared the within-position bias for each observed substitution to its substitution score in the BC0030, a substitution matrix that scores amino acid changes based on functional interchangeability (<xref ref-type="bibr" rid="B35">35</xref>), which we used as a proxy for physicochemical similarity (Figure <xref ref-type="fig" rid="F6">6</xref>C). BC0030 only accounted for &#x0007E;28% of the variation (i.e. Pearson&#x02019;s <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.28) in within-position biases (Figure <xref ref-type="fig" rid="F6">6</xref>C), suggesting that substitutions in the antibody context are not solely constrained by physicochemical properties. In addition, the observed bias is different from a previous antibody-specific amino acid substitution model (AB model) (<xref ref-type="bibr" rid="B20">20</xref>), which showed high propensities for histidine to exchange with nearly all other amino acids.</p>
<p>While the overall distribution of SHM was similar in all three healthy donors (Figure <xref ref-type="supplementary-material" rid="SM2">S2</xref>A in Supplementary Material), we found that lineages with higher SHM were more likely to contain extremely rare mutations (Figure <xref ref-type="supplementary-material" rid="SM9">S9</xref>A in Supplementary Material). Indeed, when we subsampled IGHV1-2 lineages with either low SHM (8 or fewer amino acid substitutions; 854 lineages with a combined 3,193 substitutions from germline) or high SHM (more than 20 amino acid substitutions; 132 lineages with a combined 3,136 substitutions) from the combined dataset of the three healthy donors, we found that they resulted in substantially similar profiles (Figure <xref ref-type="supplementary-material" rid="SM9">S9</xref>B in Supplementary Material) and highly correlated rarity scores for substitutions that were observed in both sets of lineages (Figure <xref ref-type="supplementary-material" rid="SM9">S9</xref>C in Supplementary Material), after accounting for the smaller number of lineages in the high SHM set. However, the high SHM lineages contained more unique substitutions that were not observed in the low SHM set (Figure <xref ref-type="supplementary-material" rid="SM9">S9</xref>D in Supplementary Material). Although this may be expected, as most possible mutations are rare, it confirms that the elicitation of antibodies with specific rare mutations is expected to require higher overall levels of SHM.</p>
</sec>
</sec>
<sec id="S3" sec-type="discussion">
<title>Discussion</title>
<p>To successfully target foreign pathogens, the immune system employs multiple mechanisms for generating diversity in the antibody repertoire, including V(D)J recombination, heavy and light chain paring, P- and N-nucleotide addition, and SHM (<xref ref-type="bibr" rid="B36">36</xref>). In humans, &#x0007E;5&#x02009;&#x000D7;&#x02009;10<sup>6</sup> new naive B cells are generated each day, and the total number of circulating B cells is &#x0007E;10<sup>11</sup> (<xref ref-type="bibr" rid="B36">36</xref>). The total number of possible distinct antibodies resulting from these processes has been estimated at up to 10<sup>18</sup> (<xref ref-type="bibr" rid="B19">19</xref>). Hence, the antibody repertoire is so diverse that a truly random search would not allow for an effective immune response in a timely fashion. Previous studies have shown that the abovementioned processes have preferences such as biased usages of V(D)J genes for recombination and biased number of nucleotides for P- and N-additions (<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B37">37</xref>&#x02013;<xref ref-type="bibr" rid="B39">39</xref>). Thus, antibodies with certain genetic signatures could occur with high probability. As a result, similar antigens can reproducibly elicit stereotypic antibodies (containing similar genetic signatures and/or somatic mutations) in many individuals [reviewed in Ref (<xref ref-type="bibr" rid="B37">37</xref>)]. GSSPs reveal that each V gene explores a unique subset of mutational space, which may limit the spectrum of antigens that can be recognized. Thus, the evolution of many divergent germline V genes represents an important strategy to expand the total mutational space sampled and to increase the number of antigens that can be recognized.</p>
<p>SHM has also long been known to operate in a biased fashion, as AID has preferences for hotspot motifs and avoids coldspot motifs (<xref ref-type="bibr" rid="B6">6</xref>&#x02013;<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B11">11</xref>&#x02013;<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B33">33</xref>). Much effort has been invested in building context-dependent models to predict the mutation propensity of Ig genes at nucleotide level (<xref ref-type="bibr" rid="B14">14</xref>&#x02013;<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B24">24</xref>). These models have emphasized the difference between the generation of mutations by SHM machinery and the selection of those mutations into the functional repertoire. This distinction is critical for insight into the biological mechanisms of SHM, as well as for a proper understanding of the functional consequences of individual mutations and how affinity toward a particular antigen is increased. Nonetheless, we show that a simple model incorporating the combined effects of both processes reveals that observed substitutions have gene- and position-specific stationary frequencies.</p>
<p>In this study, we explored the applicability of gene-specific profiles for inferring SHM propensities of V genes at the amino acid level. We demonstrated that the sampled mutational space for each V gene is strongly constrained, giving rise to substitution frequency and substitution biases that are consistent between donors and over time. Similar consistency between donors is also observed at the nucleotide level (<xref ref-type="bibr" rid="B21">21</xref>). We also demonstrated that the GSSP of IGHV3-23 functional antibodies is highly correlated with that of the passenger allele profile. Thus, the GSSPs describe the mutational space sampled by SHM machinery. Because GSSPs characterize SHM propensity at each gene position, researchers lacking experience with substitution models can still use these profiles for evaluating the propensity of amino acid mutations of interests. We also have made our scripts for building, comparing, and plotting GSSPs available for researchers to build profiles from their own datasets and use GSSPs in their studies.</p>
<p>The analysis of GSSPs advances our understanding of the SHM process. Overall, we showed that most mutations are observed only very rarely or not at all in the repertoires we examined. While the nucleotide-based S5F model that we tested was able to capture many but not all of the dominant mutations, the GSSP of the simulated repertoire displayed differences from those calculated from actual data (Figure <xref ref-type="fig" rid="F5">5</xref>A). Moreover, since the mutability of a particular nucleotide in the S5F model can change dramatically due to changes in the surrounding bases, it is difficult to calculate the likelihood of observing a specific mature antibody sequence which could have evolved through many different paths. Conversely, GSSPs effectively sum over all possible evolutionary paths and therefore provide a basis for developing methods to directly estimate the probability of observing similar SHM patterns.</p>
<p>Overall, antigen-driven selection acting on individual B cells does not appear to be a dominant factor modulating the substitutions observed in the profile of the overall repertoire. This is likely because most of the effects of antigen-driven selection are canceled out when the substitutions of many lineages are incorporated. The observation that the SHM machinery generates a limited pool of available mutations with frequencies that remain mostly unchanged after selection against varying sets of antigenic challenges in different people suggests the presence of an evolutionary equilibrium between the biases of the SHM machinery and the types and positions of mutations that are most likely to produce favorable substitutions. Indeed, previous studies have shown that codon usage within antibody V genes has likely been optimized by evolution to enhance the potential for beneficial changes via SHM (<xref ref-type="bibr" rid="B40">40</xref>). A recent study demonstrated that a substitution F83A in light chain VK1 far away from the paratope, nevertheless appeared with high frequency in mouse antibodies, improves epitope binding affinity by regulating the torsion angle of heavy-light chain pairing (<xref ref-type="bibr" rid="B31">31</xref>). Thus, investigating the functional impacts of the dominant substitutions revealed by GSSPs will be important for understanding antibody affinity maturation process and antibody design. Nonetheless, differences between substitutions in functional and passenger allele repertoires can still be observed in both this study and previous work (<xref ref-type="bibr" rid="B34">34</xref>). McCoy et al showed that approximately 30% of positions at the framework 3 region of human heavy chain genes are under significant negative selection (<xref ref-type="bibr" rid="B34">34</xref>). Although varying between genes, the positions under selection are roughly consistent between donors for each gene. Because each donor experiences different infection history, this suggests that the selection may be driven by factors other than antigen specificity. One possibility is that certain mutations are systematically and consistently filtered out by selection for structural stability (<xref ref-type="bibr" rid="B34">34</xref>). Since GSSPs provide a pool of SHMs available for antigen-driven selection, the removal of those detrimental mutations in the mutational space will not impact the general usefulness of GSSPs, but are nonetheless of biological interest. We are currently conducting investigations of the functional effects of high-frequency substitutions which will help elucidate this relationship between SHM biases and selected substitutions.</p>
<p>The GSSPs presented in this study were constructed mainly using repertoire data from 454 pyrosequencing. Although the 454 pyrosequencing technique is error-prone, most errors are indels (&#x0007E;0.09% substitution error rate vs. &#x0007E;0.9% indel error rate) (<xref ref-type="bibr" rid="B41">41</xref>), which cause frame shifts and can be easily filtered out (see <xref ref-type="sec" rid="S4">Materials and Methods</xref>). The base substitution errors on the 454 platform are similar to or lower than substitution errors observed using the Illumina MiSeq or HiSeq platforms (&#x0007E;0.25% substitution error rate) (<xref ref-type="bibr" rid="B41">41</xref>). Moreover, we showed in Figure <xref ref-type="fig" rid="F4">4</xref> that profiles from donors sequenced using both 454 and Illumina MiSeq (donor Z258) are highly similar. Thus, residual sequencing errors do not appear to affect our conclusions. Nonetheless, as demonstrated in Figure S1, more robust GSSPs can be constructed by incorporating more antibody lineages. Thus, we will incorporate more repertoire data to improve the prediction accuracy of the GSSPs in the future. We will also construct GSSPs for animal models including mouse, guinea pig, and non-human primates.</p>
<p>In this work, we have focused on substitution biases in the V genes, which comprise the bulk of the antibody variable region. However, most antibody paratopes include significant portions of the CDR3s, especially from heavy chain. Because it is difficult to assign D genes accurately and much of the CDR3 is not genetically encoded at all, the methods used in this study are not directly applicable. More accurate antibody-specific nucleotide- and amino acid substitution models that can predict the mutational space for a specific CDR3 and/or J gene region would be extremely useful and represent an important avenue for future work.</p>
<p>Gene-specific substitution profiles can aid interpretation of SHM patterns observed in mature antibodies and can shed light on their developmental pathways. The observation of dominant mutations suggests that the sampling of various combinations of the dominant mutations is an important and efficient mechanism to increase antigen specificity. This is consistent with our recent study showing that antibody clones elicited by the same immunogen in different donors share highly similar substitutions (<xref ref-type="bibr" rid="B26">26</xref>), suggesting that a dominant maturation pathway may exist. GSSPs provide a means to evaluate the probability of eliciting specific desired mutations for a target antibody modality. However, it is not yet clear if GSSPs provide a prospective mechanism for predicting likely antibody responses to specific immunogens. To address this question, a Markov Chain Monte Carlo simulation system is under development to simulate affinity maturation using GSSPs and using a structural bioinformatics module to evaluate the structural and functional effects of the introduced mutations. By simulating the maturation processes of hundreds to thousands of antibody clones, we expect to identify dominant maturation pathways for antigen-specific antibodies and to predict the likelihood of reproducing SHM patterns of interest. In the future, we will also build GSSPs for genes of animal models. We will be able to evaluate whether similar SHM patterns observed in human antibodies can be elicited in antibodies originated from homolog genes in animal models. Such analysis will be helpful for preclinical vaccine trials. Thus, GSSPs may thereby provide an important tool for rational vaccine design.</p>
</sec>
<sec id="S4" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S4-1">
<title>Donor Consent Information</title>
<p>Anonymized human PBMCs from normal, healthy donors were obtained through the NIH Clinical Center Department of Transfusion Medicine apheresis program by automated leukapheresis. Signed informed consent from the donors was obtained in accordance with the Declaration of Helsinki, and the study was approved by the National Institute of Allergy and Infectious Diseases (NIAID) Institutional Review Board. PBMCs were prepared by density gradient separation using Ficoll Paque Plus (GE HealthCare Life BioSciences, AB). Cells were then frozen in heat-inactivated fetal calf serum:DMSO (90:10) and stored at &#x02212;185&#x000B0;C until needed.</p>
</sec>
<sec id="S4-2">
<title>Next-Generation Sequencing</title>
<p>Immunoglobulin genes were amplified from PBMC samples from three HIV- and hepatitis C-negative individuals as previously described in Ref. (<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B42">42</xref>&#x02013;<xref ref-type="bibr" rid="B44">44</xref>). Briefly, human PBMCs (6&#x02009;&#x000D7;&#x02009;10<sup>7</sup>) were previously obtained from three HIV-1 and hepatitis C-negative individuals (LP32647, LP08248 and LP23810), and the PBMCs were pelleted at 1,200&#x02009;rpm for 8&#x02009;min. mRNA was then extracted and eluted in 50&#x02009;&#x000B5;l elution buffer using &#x003BC;MACS mRNA isolation kit (Miltenyi Biotec) according to the manufacturer&#x02019;s instructions. The mRNA was then aliquoted for cDNA synthesis, and a 5&#x02032;RACE approach was used to amplify IgG genes from one aliquot from each donor. The datasets were published and deposited in the NCBI Short Reads Archive (SRA ID: SRP067168) (<xref ref-type="bibr" rid="B25">25</xref>).</p>
<p>In this study, we prepared additional IgG libraries from aliquots of the same mRNA using 5&#x02032; RACE, as well as an IgG library using 5&#x02032; multiplex primers. In addition, we generated libraries for IgM, IgK, and IgL Ig genes using both 5&#x02032;RACE or 5&#x02032;multiplex methods. 5&#x02032;RACE was conducted based on previously described methods (42). Briefly, to synthesize cDNA, 10&#x02009;&#x000B5;l mRNA was mixed with 1&#x02009;&#x000B5;l 5&#x02032;CDS Oligo dT primers (12&#x02009;&#x000B5;M) and incubated at 70&#x000B0;C for 1&#x02009;min and then &#x02212;20&#x000B0;C for 1&#x02009;min. Then, 1&#x02009;&#x000B5;l SMARTER Oligo Primer (12&#x02009;&#x000B5;M) (Clontech), 4&#x02009;&#x000B5;l 5&#x000D7; RT buffer, 1&#x02009;&#x000B5;l DTT 20 (20&#x02009;mM), 1&#x02009;&#x000B5;l dNTP (10&#x02009;mM), 1&#x02009;&#x000B5;l RNAse out, and 1&#x02009;&#x000B5;l SuperScript II reverse transcriptase (Invitrogen) were added to the reaction. After 2&#x02009;h of incubation at 42&#x000B0;C, the cDNA products were purified using Nucleospin II kit (Macherey-Nagel) and eluted in 50&#x02009;&#x000B5;l water. For Ig gene recovery, 10&#x02009;&#x000B5;l cDNA was used for each PCR reaction. The first PCR amplification was performed with a common 5&#x02032; primer II A (Clontech) and Ig constant region-specific 3&#x02032; primer (IgG: 5&#x02032;GGGGAAGACCGATGGGCCCTTGGTGG3&#x02032;; IgM: 5&#x02032;GAGGGGGAAAAGGGTTGGGGCGG3&#x02032;, IgK: 5&#x02032;GGAAGATGAAGACAGATGGTGCAGCCACAG3&#x02032;, IgL: 5&#x02032;CCTTGTTGGCTTGAAGCTCCTCAGAGGAGG3&#x02032;) using KAPA HIFI qPCR kit (Kapa Biosystems). The PCR products were purified with 2% Size Select Clonewell E-gel (Invitrogen) and Agencourt AMPure XP beads (Beckman Coulter). The second PCR amplification was performed with primers with 454 sequencing adapters (454-RACE-F: 5&#x02032;CCATCTCATCCCTGCGTGTCTCCGACTCAGAAGCAGTGGTATCAACGCAGAGT3&#x02032;; 454-IgG-R: 5&#x02032;CCTATCCCCTGTGTGCCTTGGCAGTCTCAGGGGGAAGACCGATGGGCCCTTGGTGG3&#x02032;; 454-IgM-R: 5&#x02032;CCTATCCCCTGTGTGCCTTGGCAGTCTCAGGAGGGGGAAAAGGGTTGGGGCGG3&#x02032;; 454-IgK-R: 5&#x02032;CCTATCCCCTGTGTGCCTTGGCAGTCTCAGGGAAGATGAAGACAGATGGTGCAGCCACAG3&#x02032;; 454-IgL-R: 5&#x02032;CCTATCCCCTGTGTGCCTTGGCAGTCTCAGCCTTGTTGGCTTGAAGCTCCTCAGAGGAGG3&#x02032;). The PCR products were again purified with 2% Size Select Clonewell E-gel and Agencourt AMPure XP beads. For 5&#x02032; multiplex method, 10&#x02009;&#x000B5;l mRNA was used for cDNA synthesis as described previously (43). cDNA was eluted in 50&#x02009;&#x000B5;l water, and 10&#x02009;&#x000B5;l cDNA was used for each PCR reaction. As described previously (43), IgG and IgM genes were amplified using mixed primers for all VH families, while IgK and IgL genes were amplified using mixed kappa and lambda primers, respectively. PCR products were purified with 2% Size Select Clonewell E-gel and Agencourt AMPure XP beads.</p>
<p>454 pyrosequencing for all libraries was performed as described previously (<xref ref-type="bibr" rid="B45">45</xref>), the datasets were deposited to NCBI SRA database (Bioprojects: PRJNA336331). The next-generation sequencing data for the antibody repertoires of the HIV infected donors NIH45, CAP256, and Z258 were published in Ref. (<xref ref-type="bibr" rid="B44">44</xref>) (SRA ID: SRP052625) (<xref ref-type="bibr" rid="B43">43</xref>, <xref ref-type="bibr" rid="B46">46</xref>), (SRA ID: SRP034555 and SRP017087), and (Huang et al. Immunity 2016, Accepted) (SRA ID: SRR4417615-SRR4417632), respectively. The non-productive sequences derived from IGHV3-23 were retrieved from the European Nucleotide Archive (accession numbers AM076988-AM083316) (<xref ref-type="bibr" rid="B32">32</xref>).</p>
</sec>
<sec id="S4-3">
<title>Quality Control and Generation of Non-Redundant Datasets</title>
<p>In order to ensure that our calculations were not affected by sequencing error from the 454 platform, we used a stringent quality control protocol as described in the following sections. We used SONAR<xref ref-type="fn" rid="fn2"><sup>2</sup></xref> to assign germline V and J genes for each transcript (<xref ref-type="bibr" rid="B27">27</xref>). Sequences for which assignments could not be made were removed. In addition, transcripts contained stop codons or out-of-frame junctions were also removed, as these are likely to be the result of sequencing error. Because members of an expanded clonal lineage are likely to share substitutions which did not arise independently, we included only one transcript from each lineage. Lineages were found using SONAR and defined as having the same V and J gene assignments and CDR3 sequences of the same length and at least 90% nucleotide identity. Lineages containing only a single transcript were discarded, as it is impossible to verify the quality of these sequences. For all lineages containing multiple sequences, the transcript closest to the consensus sequence of the lineage was chosen as the representative, which helps minimize the effects of sequencing error (<xref ref-type="bibr" rid="B44">44</xref>). In addition, because the primary mode of 454 error is indel, the remaining transcripts were each individually aligned to the assigned germline V gene with CLUSTALW to check for frameshift error. Any sequences with non-codon-length indels were discarded. Finally, each transcript was translated and compared to the amino acid sequence of the assigned germline V gene, and any transcripts with no amino acid changes were discarded, as they do not contribute any information to the GSSP. Paired-end reads obtained from donor Z258 using the Illumina MiSeq platform were merged into a full-length amplicons using USEARCH (<xref ref-type="bibr" rid="B47">47</xref>), discarding sequences containing&#x02009;&#x0003E;&#x02009;25 mismatches in the overlap region. Merged reads with two or more expected errors (based on PHRED scores) were also discarded, and the remaining reads were processed in the fashion described earlier. Overall, we processed nearly 38 million NGS reads across 54 samples from six donors, which resulted in 14.6 million high-quality reads and over 200,000 lineages which we used to construct GSSPs (Table S1).</p>
</sec>
<sec id="S4-4">
<title>Effects of Selecting a Representative Sequence from Each Lineage</title>
<p>Donor LP08248 had 284 VH3-23-derived lineages with at least 10 member sequences. We conducted 100 trials in which a random sequence was chosen as the representative of each lineage, and a mutability profile was constructed as described in the following sections. For each of these profiles, we calculated a set of rarity scores. We then calculated the Pearson&#x02019;s correlation coefficient between the rarity scores of each pair of profiles; the distribution of these coefficients is shown in Figure S1B.</p>
</sec>
<sec id="S4-5">
<title>Single Lineage Rarity Scores</title>
<p>We retrieved 348 VH3-30-derived sequences from the CAP256-VRC26 broadly HIV-1-neutralizing antibody lineage (<xref ref-type="bibr" rid="B46">46</xref>), including 33 experimentally isolated monoclonal antibodies. We constructed a GSSP from these sequences and compared the rarity scores derived from this profile to rarity scores derived from a VH3-30 profile constructed from the merged lineages of all three HIV<sup>&#x02212;</sup> donors (Figures <xref ref-type="supplementary-material" rid="SM6">S6</xref>C,D in Supplementary Material). For comparison, we also constructed a profile from 378 VH3-23 derived sequences from a single lineage found in donor LP08248 (Figures <xref ref-type="supplementary-material" rid="SM6">S6</xref>A,B in Supplementary Material).</p>
</sec>
<sec id="S4-6">
<title>Identification of Potential Novel Germline Alleles</title>
<p>Identifying mutations requires an accurate knowledge of the germline sequence. Undocumented polymorphisms can appear as donor-specific, high-frequency mutations, skewing the resulting GSSPs and exaggerating differences between donors. We therefore used partis (<xref ref-type="bibr" rid="B21">21</xref>) to infer novel germline alleles from each non-redundant dataset. (Allele finding is considered a beta feature, available from <uri xlink:href="https://github.com/psathyrella/partis">https://github.com/psathyrella/partis</uri>. We used commit <monospace>610ae46</monospace> from June 18<sup>th</sup>, 2016.) Across all 6 donors, we found 33 unique novel heavy chain alleles, 23 novel kappa chain alleles, and 9 novel lambda chain alleles. Of these, 3, 5, and 4, respectively, were observed in two or more of these donors. The alleles have been deposited in the IgPdb<xref ref-type="fn" rid="fn3"><sup>3</sup></xref>.</p>
</sec>
<sec id="S4-7">
<title>Construction and Comparison of V Gene Substitution Profiles</title>
<p>For each non-redundant dataset, we used CLUSTALW2 to align translated transcripts from each V gene to all alleles of that gene, including novel alleles detected by partis. For each position, we calculated the substitution frequency and rarity as described in the text. For polymorphic positions, all possible polymorphisms were considered germline residues, without confirming the presence of each allele in each donor. Gaps and undefined amino acids were excluded from counting. GSSPs were only calculated if there were at least 100 sequences for a particular V gene in the given dataset. When building GSSPs, IMGT alleles IGHV4-4&#x0002A;07 and IGHV4-4&#x0002A;08 were considered as alleles of the IGHV4-59 gene, due to sequence similarity. Logo plots of GSSPs were shown using a customized version of WebLogo (<xref ref-type="bibr" rid="B48">48</xref>).</p>
<p>To compare two GSSPs, we calculated the Jensen&#x02013;Shannon divergence between the distributions of mutations or substitutions in the two profiles at each position in the V gene. This is defined as
<disp-formula id="E2"><mml:math id="M2"><mml:mrow><mml:mi>J</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mi>H</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mn>2</mml:mn></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mrow><mml:mi>H</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:mi>H</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mfrac></mml:mrow></mml:math></disp-formula>
where <italic>P<sub>A,i</sub></italic> and <italic>P<sub>B,i</sub></italic> are vectors of length 20 describing the probability of observing each possible non-germline residue at position <italic>i</italic> for profiles <italic>A</italic> and <italic>B</italic>, respectively. <italic>H</italic> is the Shannon entropy, defined as
<disp-formula id="E3"><mml:math id="M3"><mml:mrow><mml:mi>H</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mo>&#x0007B;</mml:mo><mml:mi>a</mml:mi><mml:mi>a</mml:mi><mml:mo>&#x0007D;</mml:mo></mml:mrow></mml:munder><mml:mrow><mml:msubsup><mml:mi>P</mml:mi><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:msubsup></mml:mrow></mml:mstyle><mml:msub><mml:mrow><mml:mtext>log</mml:mtext></mml:mrow><mml:mn>2</mml:mn></mml:msub><mml:msubsup><mml:mi>P</mml:mi><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:msubsup></mml:mrow></mml:math></disp-formula>
where <italic>j</italic> can be any of the 20 amino acids. These position-wise Jensen&#x02013;Shannon divergences were then averaged, with the divergence of each position weighted by the mean substitution frequency for that position in the two profiles being compared:
<disp-formula id="E4"><mml:math id="M4"><mml:mrow><mml:mrow><mml:mo>&#x02329;</mml:mo><mml:mrow><mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>B</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x0232A;</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mn>2</mml:mn></mml:mfrac></mml:mrow></mml:mstyle><mml:mi>J</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle='true'><mml:msub><mml:mo>&#x02211;</mml:mo><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mn>2</mml:mn></mml:mfrac></mml:mrow></mml:mstyle></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula>
where <italic>m<sub>A,i</sub></italic> and <italic>m<sub>B,i</sub></italic> are the substitution frequency (observed frequency of mutation) at position <italic>i</italic> for profiles <italic>A</italic> and <italic>B</italic>, respectively. For comparisons of GSSPs from different V genes, only homologous positions were used. We generated a matrix of divergences among the datasets studied and used the <monospace>cmdscale</monospace> command in R to do multidimensional scaling and generate coordinates for plotting.</p>
</sec>
<sec id="S4-8">
<title>Passenger Allele GSSP</title>
<p>The non-productive sequences were manually aligned to germline gene IGHV3-23 to remove indels in order to produce a meaningful GSSP. In cases where the boundaries of an indel were ambiguous, the boundary bases were replaced with Ns to avoid potential bias.</p>
</sec>
<sec id="S4-9">
<title>Antibody Somatic Mutation Simulation</title>
<p>Starting from the nucleotide sequences of IGHV3-23&#x0002A;01, the S5F model was used to generate an artificial repertoire of 420 lineages, each with an independently simulated set of mutations. We produced 15 sequences each with 1&#x02013;28 nucleotide changes; after accounting for the possibilities of silent mutations or multiple changes in a single codon, this produced a distribution of amino acid changes that roughly matched observed distributions of SHM in the functional repertoire (Figure <xref ref-type="supplementary-material" rid="SM5">S5</xref>C in Supplementary Material). For each sequence, we calculated the mutability of each nucleotide position under the S5F model and randomly selected a site for mutation based on those probabilities. The resulting substitution was then randomly chosen based on the substitution biases indicated for the target base by the S5F model. We excluded mutations that resulted in stop codons, but did not exclude re-mutation at a previously mutated site. For simulations with the AB model, positions to be mutated were selected using our own observations of mutation frequencies as calculated for the GSSP of IGHV3-23 and the substitution was then chosen under the AB model.</p>
</sec>
<sec id="S4-10">
<title>Average Substitution Frequencies of the 20 Amino Acids</title>
<p>The GSSPs of 69&#x02009;V genes (26 heavy chain, 20 kappa chain, and 23 lambda chain) were used to calculate the substitution frequencies. Because the substitution frequency at positions with low substitution frequency may be undersampled, we excluded positions being mutated in less than 5% of the V gene-specific clones. We also excluded positions containing amino acid changes among alleles. For each selected position of a profile, we normalized the substitution frequency to 1. Finally, all selected positions were grouped based on germline amino acid type. For each of the 20 amino acids, the substitution frequency to each of the other amino acids was averaged over all positions within a group.</p>
</sec>
<sec id="S4-11">
<title>Statistical Test</title>
<p>The Mann&#x02013;Whitney <italic>U</italic> test was used to evaluate the similarities of features between different groups in this study. Pearson&#x02019;s correlations and statistical significance were calculated using <monospace>cor.test</monospace> in R.</p>
</sec>
</sec>
<sec id="S5">
<title>Ethics Statement</title>
<p>All work related to human subjects was performed in compliance with protocols approved by the NIH Institutional Review Board.</p>
</sec>
<sec id="S6" sec-type="author-contributor">
<title>Author Contributions</title>
<p>ZS, CS, PK, and LS designed research; ZS, CS, RK, JM, and LS performed research; ZS and CS analyzed data; ZS, CS, RK, JM, PK, and LS wrote the paper. All authors reviewed, commented on, and approved the manuscript.</p>
</sec>
<sec id="S7">
<title>Consortia</title>
<p>The NISC Comparative Sequencing Program includes Betty Benjamin, Gerry Bouffard, Shelise Brooks, Holly Coleman, Mila Dekhtyar, Xiaobin Guan, Joel Han, Shi- ling Ho, Richelle Legaspi, Quino Maduro, Cathy Masiello, Jenny McDowell, Casandra Montemayor, James Mullikin, Morgan Park, Nancy Riebow, Jessica Rosarda, Karen Schandler, Brian Schmidt, Christina Sison, Ray Smith, Mal Stantripop, James Thomas, Pam Thomas, Meg Vemulapalli, and Alice Young.</p>
</sec>
<sec id="S8">
<title>Conflict of Interest Statement</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest. The reviewer, PG, declared a shared affiliation, though no other collaboration, with the authors to the handling Editor, who ensured that the process nevertheless met the standards of a fair and objective review.</p>
</sec>
</body>
<back>
<sec id="S9">
<title>Funding</title>
<p>Funding was provided in part by the intramural program of the Vaccine Research Center, National Institute of Allergy and Infectious Disease, National Institutes of Health. Funding was also provided by HIVRAD grant AI104722-3 and R01 AI104387-3 to LS. Computational resources used for the described analyses were supported in part by NIH instrumentation grants S10OD012351 and S10OD021764 to the Department of Systems Biology at Columbia University.</p>
</sec>
<sec id="S10" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at <uri xlink:href="http://journal.frontiersin.org/article/10.3389/fimmu.2017.00537/full&#x00023;supplementary-material">http://journal.frontiersin.org/article/10.3389/fimmu.2017.00537/full&#x00023;supplementary-material</uri>.</p>
<supplementary-material xlink:href="Table_1.XLS" id="SM10" mimetype="applicationn/XLS" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Presentation_1.PDF" id="SM1" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Figure S1</label>
<caption><p><bold>Effects of sampling size on the robustness of GSSPs</bold>. <bold>(A)</bold> For each of eight common VH genes, GSSPs were built using randomly sampled sets of 25, 50, 100, 200, or 300 clonal lineages. The Jensen&#x02013;Shannon divergence was calculated between profiles from different donors but from the same gene and using the same number of lineages. Larger datasets resulted in lower Jensen&#x02013;Shannon divergences, but the divergences converged when GSSPs were built using &#x0007E;300 lineages, suggesting 300 lineages are required to build a robust GSSP. <bold>(B)</bold> To estimate the effects of the choice of representative sequence on GSSP construction, we randomly selected one representative sequence per lineage and built 100 repertoires. The rarity scores of mutations of randomly resampled repertoires showed an average correlation coefficient of &#x0007E;0.98, suggesting that the rarity of mutations is robust to choice of representative sequences.</p></caption></supplementary-material>
<supplementary-material xlink:href="Presentation_1.PDF" id="SM2" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Figure S2</label>
<caption><p><bold>Comparisons of the GSSPs of IGHV1-2 and IGHV1-69 and the distributions of SHM of IGHV1-2, IGHV1-69, and IGKV3-20 among three healthy donors</bold>. <bold>(A)</bold> The distributions of amino acid SHM levels are similar among the three donors, as exemplified by IGHV1-2, IGHV1-69, and IGKV3-20. The Pearson correlation coefficients are listed, the correlations of which are all significant (<italic>p</italic>&#x02009;&#x0003C;&#x02009;0.01). <bold>(B)</bold> The GSSP of IGHV1-2 constructed using lineages from the three healthy donors. <bold>(C)</bold> The GSSP of IGHV1-69 built using lineages from each of the three donors is noticeably different from those of IGHV1-2.</p></caption></supplementary-material>
<supplementary-material xlink:href="Presentation_1.PDF" id="SM3" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Figure S3</label>
<caption><p><bold>GSSPs of codon AAG and CTG in various nucleotide contexts and correlation between substitution rarity and number of mutations within a codon</bold>. <bold>(A)</bold> The GSSPs of codon AAG (encoding Lysine) are more similar at homologous positions (similar nucleotide context) than between non-homologous positions. <bold>(B)</bold> The GSSPs of codon CTG (encoding Leucine) are more similar at homologous positions than between non-homologous positions. Color scheme: green, amino acid replacement involves single nucleotide-mutation; yellow, amino acid replacement involves two nucleotide-mutations; red, amino acid replacement involves three nucleotide-mutations. <bold>(C)</bold> Amino acid mutations requiring more nucleotide substitutions within a codon tend to be rarer, partially explaining why many mutations are rare.</p></caption></supplementary-material>
<supplementary-material xlink:href="Presentation_1.PDF" id="SM4" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Figure S4</label>
<caption><p><bold>Similarities of the GSSPs of V genes and over time</bold>. For each of the three V gene types, <bold>(A)</bold> heavy chain, <bold>(B)</bold> kappa chain, and <bold>(C)</bold> labmda chain, the similarities measured by the Jensen&#x02013;Shannon divergence and visualized using multidimensional scaling showed that the GSSPs of each V gene is similar among the three healthy donors, and V genes of the same family showed more similar GSSPs than between V families. We also observed that the GSSPs of a V gene are highly consistent over time for both <bold>(D)</bold> heavy chain and <bold>(E)</bold> kappa chain in donor NIH45.</p></caption></supplementary-material>
<supplementary-material xlink:href="Presentation_1.PDF" id="SM5" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Figure S5</label>
<caption><p><bold>Similarities of substitution frequency and substitution bias among three healthy donors</bold>. <bold>(A)</bold> The overall substitution frequency of all V genes are highly consistent between the three donors, indicating the substitution frequency of V gene is consistently modulated. <bold>(B)</bold> The overall rarity scores observed in all V genes are highly consistent among the three donors, suggesting the substitution bias is also consistently modulated. <bold>(C)</bold> Distributions of somatic hypermutation levels for VH3-23 lineages in the functional, passenger allele, and simulated repertoires shown in Figure <xref ref-type="fig" rid="F4">4</xref>.</p></caption></supplementary-material>
<supplementary-material xlink:href="Presentation_1.PDF" id="SM6" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Figure S6</label>
<caption><p><bold>Similarity of rarity scores of 08248-00037 and CAP256-VRC26 lineages to the combined repertoire of the three healthy donors</bold>. <bold>(A)</bold> The GSSP of the VH3-23-derived lineage 0037 from donor 08248 (bottom) compared to the overall GSSP of IGHV3-23 from the functional and passenger allele repertoires. <bold>(B)</bold> The correlation of rarity score between 08248-00037 and the overall GSSP of IGHV3-23 from the functional repertoire is non-significant. <bold>(C,D)</bold> The same plots for the VH3-30-derived broadly HIV-1-neutralizing lVRC26 lineage from donor CAP256 compared to the overall GSSP of IGHV3-30 from functional repertoires.</p></caption></supplementary-material>
<supplementary-material xlink:href="Presentation_1.PDF" id="SM7" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Figure S7</label>
<caption><p><bold>Similarity of substitution frequencies and rarities at sites in the framework 3 region of IGHV3-23 compared by selection type detected by McCoy et al</bold>. <bold>(A)</bold> Position-specific substitution rates calculated by McCoy et al were downloaded from doi:<uri xlink:href="https://doi.org/10.6084/m9.figshare.1399201">10.6084/m9.figshare.1399201</uri> and rows containing &#x0201C;dNdS&#x0201D; estimates for the normalized &#x0201C;productive/out-of-frame&#x0201D; subset were extracted. As in Ref. (<xref ref-type="bibr" rid="B34">34</xref>), we considered only rows for which both &#x0201C;coverage&#x0201D; and &#x0201C;oof_coverage&#x0201D; were greater than 100. Sites with &#x0201C;hpdUpper&#x0201D; less than 1 were classified as being under negative selection and those with &#x0201C;hpdLower&#x0201D; greater than 1 were classified as being under positive selection. Sites under negative or positive selection showed a lowered correlation of substitution frequency between functional and non-productive repertoires, reflective of modulation by selection. <bold>(B)</bold> Sites classified as being under positive selection showed a significantly lowered correlation between rarity scores in functional and non-productive repertoires, suggesting selection modulates the substitution bias observed in GSSPs. Surprisingly, sites under negative selection showed increased correlation, possibly reflecting structural effects that SHM has evolved to avoid equally in functional and non-productive sequences.</p></caption></supplementary-material>
<supplementary-material xlink:href="Presentation_1.PDF" id="SM8" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Figure S8</label>
<caption><p><bold>AB model compared to real data</bold>. <bold>(A)</bold> The GSSP of the VH3-23 repertoire as simulated under the AB model (bottom) compared to the overall GSSP of IGHV3-23 from the functional and passenger allele repertoires. <bold>(B)</bold> The correlation of rarity scores between AB-simulated repertoires and the functional repertoire of donor 08248 (left) or non-productive passenger alleles (right).</p></caption></supplementary-material>
<supplementary-material xlink:href="Presentation_1.PDF" id="SM9" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink">
<label>Figure S9</label>
<caption><p><bold>Correlations between substitution rarity and somatic hypermutation level</bold>. <bold>(A)</bold> For each chain type, all transcripts in the three healthy donors were sorted into three groups based on SHM level, and the portions of rare mutations was estimated. The analysis showed that lineages containing higher SHM levels tend to have more rare mutation. <bold>(B)</bold> The substitution biases in the GSSP of low-SHM-lineages are similar to those of high SHM lineages (note different scales for <italic>y</italic>-axis). <bold>(C)</bold> Rarity for high- and low SHM lineages are highly correlated, though rarity calculated from low SHM sequences is uniformly higher, due to the lower overall substitution frequency. <bold>(D)</bold> The mutational space sampled by low-SHM-lineages is smaller but consistent with that of high-SHM-lineages.</p></caption></supplementary-material></sec>
<ref-list>
<title>References</title>
<ref id="B1"><label>1</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Muramatsu</surname> <given-names>M</given-names></name> <name><surname>Kinoshita</surname> <given-names>K</given-names></name> <name><surname>Fagarasan</surname> <given-names>S</given-names></name> <name><surname>Yamada</surname> <given-names>S</given-names></name> <name><surname>Shinkai</surname> <given-names>Y</given-names></name> <name><surname>Honjo</surname> <given-names>T</given-names></name></person-group>. <article-title>Class switch recombination and hypermutation require activation-induced cytidine deaminase (AID), a potential RNA editing enzyme</article-title>. <source>Cell</source> (<year>2000</year>) <volume>102</volume>(<issue>5</issue>):<fpage>553</fpage>&#x02013;<lpage>63</lpage>.<pub-id pub-id-type="doi">10.1016/S0092-8674(00)00078-7</pub-id><pub-id pub-id-type="pmid">11007474</pub-id></citation></ref>
<ref id="B2"><label>2</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wood</surname> <given-names>GS</given-names></name></person-group>. <article-title>The immunohistology of lymph nodes in HIV infection: a review</article-title>. <source>Prog AIDS Pathol</source> (<year>1990</year>) <volume>2</volume>:<fpage>25</fpage>&#x02013;<lpage>32</lpage>.<pub-id pub-id-type="pmid">2103863</pub-id></citation></ref>
<ref id="B3"><label>3</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>M</given-names></name> <name><surname>Duke</surname> <given-names>JL</given-names></name> <name><surname>Richter</surname> <given-names>DJ</given-names></name> <name><surname>Vinuesa</surname> <given-names>CG</given-names></name> <name><surname>Goodnow</surname> <given-names>CC</given-names></name> <name><surname>Kleinstein</surname> <given-names>SH</given-names></name> <etal/></person-group> <article-title>Two levels of protection for the B cell genome during somatic hypermutation</article-title>. <source>Nature</source> (<year>2008</year>) <volume>451</volume>(<issue>7180</issue>):<fpage>841</fpage>&#x02013;<lpage>5</lpage>.<pub-id pub-id-type="doi">10.1038/nature06547</pub-id><pub-id pub-id-type="pmid">18273020</pub-id></citation></ref>
<ref id="B4"><label>4</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Di Noia</surname> <given-names>JM</given-names></name> <name><surname>Neuberger</surname> <given-names>MS</given-names></name></person-group>. <article-title>Molecular mechanisms of antibody somatic hypermutation</article-title>. <source>Annu Rev Biochem</source> (<year>2007</year>) <volume>76</volume>:<fpage>1</fpage>&#x02013;<lpage>22</lpage>.<pub-id pub-id-type="doi">10.1146/annurev.biochem.76.061705.090740</pub-id><pub-id pub-id-type="pmid">17328676</pub-id></citation></ref>
<ref id="B5"><label>5</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chandra</surname> <given-names>V</given-names></name> <name><surname>Bortnick</surname> <given-names>A</given-names></name> <name><surname>Murre</surname> <given-names>C</given-names></name></person-group>. <article-title>AID targeting: old mysteries and new challenges</article-title>. <source>Trends Immunol</source> (<year>2015</year>) <volume>36</volume>(<issue>9</issue>):<fpage>527</fpage>&#x02013;<lpage>35</lpage>.<pub-id pub-id-type="doi">10.1016/j.it.2015.07.003</pub-id><pub-id pub-id-type="pmid">26254147</pub-id></citation></ref>
<ref id="B6"><label>6</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Odegard</surname> <given-names>VH</given-names></name> <name><surname>Schatz</surname> <given-names>DG</given-names></name></person-group>. <article-title>Targeting of somatic hypermutation</article-title>. <source>Nat Rev Immunol</source> (<year>2006</year>) <volume>6</volume>(<issue>8</issue>):<fpage>573</fpage>&#x02013;<lpage>83</lpage>.<pub-id pub-id-type="doi">10.1038/nri1896</pub-id><pub-id pub-id-type="pmid">16868548</pub-id></citation></ref>
<ref id="B7"><label>7</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cowell</surname> <given-names>LG</given-names></name> <name><surname>Kepler</surname> <given-names>TB</given-names></name></person-group>. <article-title>The nucleotide-replacement spectrum under somatic hypermutation exhibits microsequence dependence that is strand-symmetric and distinct from that under germline mutation</article-title>. <source>J Immunol</source> (<year>2000</year>) <volume>164</volume>(<issue>4</issue>):<fpage>1971</fpage>&#x02013;<lpage>6</lpage>.<pub-id pub-id-type="doi">10.4049/jimmunol.164.4.1971</pub-id><pub-id pub-id-type="pmid">10657647</pub-id></citation></ref>
<ref id="B8"><label>8</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Betz</surname> <given-names>AG</given-names></name> <name><surname>Rada</surname> <given-names>C</given-names></name> <name><surname>Pannell</surname> <given-names>R</given-names></name> <name><surname>Milstein</surname> <given-names>C</given-names></name> <name><surname>Neuberger</surname> <given-names>MS</given-names></name></person-group>. <article-title>Passenger transgenes reveal intrinsic specificity of the antibody hypermutation mechanism: clustering, polarity, and specific hot spots</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>1993</year>) <volume>90</volume>(<issue>6</issue>):<fpage>2385</fpage>&#x02013;<lpage>8</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.90.6.2385</pub-id><pub-id pub-id-type="pmid">8460148</pub-id></citation></ref>
<ref id="B9"><label>9</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goyenechea</surname> <given-names>B</given-names></name> <name><surname>Milstein</surname> <given-names>C</given-names></name></person-group>. <article-title>Modifying the sequence of an immunoglobulin V-gene alters the resulting pattern of hypermutation</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>1996</year>) <volume>93</volume>(<issue>24</issue>):<fpage>13979</fpage>&#x02013;<lpage>84</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.93.24.13979</pub-id><pub-id pub-id-type="pmid">8943046</pub-id></citation></ref>
<ref id="B10"><label>10</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wilson</surname> <given-names>TM</given-names></name> <name><surname>Vaisman</surname> <given-names>A</given-names></name> <name><surname>Martomo</surname> <given-names>SA</given-names></name> <name><surname>Sullivan</surname> <given-names>P</given-names></name> <name><surname>Lan</surname> <given-names>L</given-names></name> <name><surname>Hanaoka</surname> <given-names>F</given-names></name> <etal/></person-group> <article-title>MSH2-MSH6 stimulates DNA polymerase eta, suggesting a role for A:T mutations in antibody genes</article-title>. <source>J Exp Med</source> (<year>2005</year>) <volume>201</volume>(<issue>4</issue>):<fpage>637</fpage>&#x02013;<lpage>45</lpage>.<pub-id pub-id-type="doi">10.1084/jem.20042066</pub-id><pub-id pub-id-type="pmid">15710654</pub-id></citation></ref>
<ref id="B11"><label>11</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rogozin</surname> <given-names>IB</given-names></name> <name><surname>Kolchanov</surname> <given-names>NA</given-names></name></person-group>. <article-title>Somatic hypermutagenesis in immunoglobulin genes. II. Influence of neighbouring base sequences on mutagenesis</article-title>. <source>Biochim Biophys Acta</source> (<year>1992</year>) <volume>1171</volume>(<issue>1</issue>):<fpage>11</fpage>&#x02013;<lpage>8</lpage>.<pub-id pub-id-type="doi">10.1016/0167-4781(92)90134-L</pub-id><pub-id pub-id-type="pmid">1420357</pub-id></citation></ref>
<ref id="B12"><label>12</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yeap</surname> <given-names>LS</given-names></name> <name><surname>Hwang</surname> <given-names>JK</given-names></name> <name><surname>Du</surname> <given-names>Z</given-names></name> <name><surname>Meyers</surname> <given-names>RM</given-names></name> <name><surname>Meng</surname> <given-names>FL</given-names></name> <name><surname>Jakubauskaite</surname> <given-names>A</given-names></name> <etal/></person-group> <article-title>Sequence-intrinsic mechanisms that target AID mutational outcomes on antibody genes</article-title>. <source>Cell</source> (<year>2015</year>) <volume>163</volume>(<issue>5</issue>):<fpage>1124</fpage>&#x02013;<lpage>37</lpage>.<pub-id pub-id-type="doi">10.1016/j.cell.2015.10.042</pub-id><pub-id pub-id-type="pmid">26582132</pub-id></citation></ref>
<ref id="B13"><label>13</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pham</surname> <given-names>P</given-names></name> <name><surname>Bransteitter</surname> <given-names>R</given-names></name> <name><surname>Petruska</surname> <given-names>J</given-names></name> <name><surname>Goodman</surname> <given-names>MF</given-names></name></person-group>. <article-title>Processive AID-catalysed cytosine deamination on single-stranded DNA simulates somatic hypermutation</article-title>. <source>Nature</source> (<year>2003</year>) <volume>424</volume>(<issue>6944</issue>):<fpage>103</fpage>&#x02013;<lpage>7</lpage>.<pub-id pub-id-type="doi">10.1038/nature01760</pub-id><pub-id pub-id-type="pmid">12819663</pub-id></citation></ref>
<ref id="B14"><label>14</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shapiro</surname> <given-names>GS</given-names></name> <name><surname>Aviszus</surname> <given-names>K</given-names></name> <name><surname>Murphy</surname> <given-names>J</given-names></name> <name><surname>Wysocki</surname> <given-names>LJ</given-names></name></person-group>. <article-title>Evolution of Ig DNA sequence to target specific base positions within codons for somatic hypermutation</article-title>. <source>J Immunol</source> (<year>2002</year>) <volume>168</volume>(<issue>5</issue>):<fpage>2302</fpage>&#x02013;<lpage>6</lpage>.<pub-id pub-id-type="doi">10.4049/jimmunol.168.5.2302</pub-id><pub-id pub-id-type="pmid">11859119</pub-id></citation></ref>
<ref id="B15"><label>15</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shapiro</surname> <given-names>GS</given-names></name> <name><surname>Ellison</surname> <given-names>MC</given-names></name> <name><surname>Wysocki</surname> <given-names>LJ</given-names></name></person-group>. <article-title>Sequence-specific targeting of two bases on both DNA strands by the somatic hypermutation mechanism</article-title>. <source>Mol Immunol</source> (<year>2003</year>) <volume>40</volume>(<issue>5</issue>):<fpage>287</fpage>&#x02013;<lpage>95</lpage>.<pub-id pub-id-type="doi">10.1016/S0161-5890(03)00101-9</pub-id><pub-id pub-id-type="pmid">12943801</pub-id></citation></ref>
<ref id="B16"><label>16</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>DS</given-names></name> <name><surname>Creadon</surname> <given-names>G</given-names></name> <name><surname>Jena</surname> <given-names>PK</given-names></name> <name><surname>Portanova</surname> <given-names>JP</given-names></name> <name><surname>Kotzin</surname> <given-names>BL</given-names></name> <name><surname>Wysocki</surname> <given-names>LJ</given-names></name></person-group>. <article-title>Di- and trinucleotide target preferences of somatic mutagenesis in normal and autoreactive B cells</article-title>. <source>J Immunol</source> (<year>1996</year>) <volume>156</volume>(<issue>7</issue>):<fpage>2642</fpage>&#x02013;<lpage>52</lpage>.<pub-id pub-id-type="pmid">8786330</pub-id></citation></ref>
<ref id="B17"><label>17</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cohen</surname> <given-names>RM</given-names></name> <name><surname>Kleinstein</surname> <given-names>SH</given-names></name> <name><surname>Louzoun</surname> <given-names>Y</given-names></name></person-group>. <article-title>Somatic hypermutation targeting is influenced by location within the immunoglobulin V region</article-title>. <source>Mol Immunol</source> (<year>2011</year>) <volume>48</volume>(<issue>12&#x02013;13</issue>):<fpage>1477</fpage>&#x02013;<lpage>83</lpage>.<pub-id pub-id-type="doi">10.1016/j.molimm.2011.04.002</pub-id><pub-id pub-id-type="pmid">21592579</pub-id></citation></ref>
<ref id="B18"><label>18</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yaari</surname> <given-names>G</given-names></name> <name><surname>Vander Heiden</surname> <given-names>JA</given-names></name> <name><surname>Uduman</surname> <given-names>M</given-names></name> <name><surname>Gadala-Maria</surname> <given-names>D</given-names></name> <name><surname>Gupta</surname> <given-names>N</given-names></name> <name><surname>Stern</surname> <given-names>JN</given-names></name> <etal/></person-group> <article-title>Models of somatic hypermutation targeting and substitution based on synonymous mutations from high-throughput immunoglobulin sequencing data</article-title>. <source>Front Immunol</source> (<year>2013</year>) <volume>4</volume>:<fpage>358</fpage>.<pub-id pub-id-type="doi">10.3389/fimmu.2013.00358</pub-id><pub-id pub-id-type="pmid">24298272</pub-id></citation></ref>
<ref id="B19"><label>19</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elhanati</surname> <given-names>Y</given-names></name> <name><surname>Sethna</surname> <given-names>Z</given-names></name> <name><surname>Marcou</surname> <given-names>Q</given-names></name> <name><surname>Callan</surname> <given-names>CG</given-names> <suffix>Jr</suffix></name> <name><surname>Mora</surname> <given-names>T</given-names></name> <name><surname>Walczak</surname> <given-names>AM</given-names></name></person-group>. <article-title>Inferring processes underlying B-cell repertoire diversity</article-title>. <source>Philos Trans R Soc Lond B Biol Sci</source> (<year>2015</year>) <volume>370</volume>(<issue>1676</issue>):<lpage>20140243</lpage>.<pub-id pub-id-type="doi">10.1098/rstb.2014.0243</pub-id><pub-id pub-id-type="pmid">26194757</pub-id></citation></ref>
<ref id="B20"><label>20</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mirsky</surname> <given-names>A</given-names></name> <name><surname>Kazandjian</surname> <given-names>L</given-names></name> <name><surname>Anisimova</surname> <given-names>M</given-names></name></person-group>. <article-title>Antibody-specific model of amino acid substitution for immunological inferences from alignments of antibody sequences</article-title>. <source>Mol Biol Evol</source> (<year>2015</year>) <volume>32</volume>(<issue>3</issue>):<fpage>806</fpage>&#x02013;<lpage>19</lpage>.<pub-id pub-id-type="doi">10.1093/molbev/msu340</pub-id><pub-id pub-id-type="pmid">25534034</pub-id></citation></ref>
<ref id="B21"><label>21</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ralph</surname> <given-names>DK</given-names></name> <name><surname>Matsen</surname> <given-names>FA</given-names> <suffix>IV</suffix></name></person-group>. <article-title>Consistency of VDJ rearrangement and substitution parameters enables accurate B cell receptor sequence annotation</article-title>. <source>PLoS Comput Biol</source> (<year>2016</year>) <volume>12</volume>(<issue>1</issue>):<fpage>e1004409</fpage>.<pub-id pub-id-type="doi">10.1371/journal.pcbi.1004409</pub-id><pub-id pub-id-type="pmid">26751373</pub-id></citation></ref>
<ref id="B22"><label>22</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Uduman</surname> <given-names>M</given-names></name> <name><surname>Shlomchik</surname> <given-names>MJ</given-names></name> <name><surname>Vigneault</surname> <given-names>F</given-names></name> <name><surname>Church</surname> <given-names>GM</given-names></name> <name><surname>Kleinstein</surname> <given-names>SH</given-names></name></person-group>. <article-title>Integrating B cell lineage information into statistical tests for detecting selection in Ig sequences</article-title>. <source>J Immunol</source> (<year>2014</year>) <volume>192</volume>(<issue>3</issue>):<fpage>867</fpage>&#x02013;<lpage>74</lpage>.<pub-id pub-id-type="doi">10.4049/jimmunol.1301551</pub-id><pub-id pub-id-type="pmid">24376267</pub-id></citation></ref>
<ref id="B23"><label>23</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sheng</surname> <given-names>Z</given-names></name> <name><surname>Schramm</surname> <given-names>CA</given-names></name> <name><surname>Connors</surname> <given-names>M</given-names></name> <name><surname>Morris</surname> <given-names>L</given-names></name> <name><surname>Mascola</surname> <given-names>JR</given-names></name> <name><surname>Kwong</surname> <given-names>PD</given-names></name> <etal/></person-group> <article-title>Effects of darwinian selection and mutability on rate of broadly neutralizing antibody evolution during HIV-1 infection</article-title>. <source>PLoS Comput Biol</source> (<year>2016</year>) <volume>12</volume>(<issue>5</issue>):<fpage>e1004940</fpage>.<pub-id pub-id-type="doi">10.1371/journal.pcbi.1004940</pub-id><pub-id pub-id-type="pmid">27191167</pub-id></citation></ref>
<ref id="B24"><label>24</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cui</surname> <given-names>A</given-names></name> <name><surname>Di Niro</surname> <given-names>R</given-names></name> <name><surname>Vander Heiden</surname> <given-names>JA</given-names></name> <name><surname>Briggs</surname> <given-names>AW</given-names></name> <name><surname>Adams</surname> <given-names>K</given-names></name> <name><surname>Gilbert</surname> <given-names>T</given-names></name> <etal/></person-group> <article-title>A model of somatic hypermutation targeting in mice based on high-throughput Ig sequencing data</article-title>. <source>J Immunol</source> (<year>2016</year>) <volume>197</volume>(<issue>9</issue>):<fpage>3566</fpage>&#x02013;<lpage>74</lpage>.<pub-id pub-id-type="doi">10.4049/jimmunol.1502263</pub-id><pub-id pub-id-type="pmid">27707999</pub-id></citation></ref>
<ref id="B25"><label>25</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bonsignori</surname> <given-names>M</given-names></name> <name><surname>Zhou</surname> <given-names>T</given-names></name> <name><surname>Sheng</surname> <given-names>Z</given-names></name> <name><surname>Chen</surname> <given-names>L</given-names></name> <name><surname>Gao</surname> <given-names>F</given-names></name> <name><surname>Joyce</surname> <given-names>MG</given-names></name> <etal/></person-group> <article-title>Maturation pathway from germline to broad HIV-1 neutralizer of a CD4-mimic antibody</article-title>. <source>Cell</source> (<year>2016</year>) <volume>165</volume>(<issue>2</issue>):<fpage>449</fpage>&#x02013;<lpage>63</lpage>.<pub-id pub-id-type="doi">10.1016/j.cell.2016.02.022</pub-id><pub-id pub-id-type="pmid">26949186</pub-id></citation></ref>
<ref id="B26"><label>26</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>M</given-names></name> <name><surname>Cheng</surname> <given-names>C</given-names></name> <name><surname>Chen</surname> <given-names>X</given-names></name> <name><surname>Duan</surname> <given-names>H</given-names></name> <name><surname>Cheng</surname> <given-names>HL</given-names></name> <name><surname>Dao</surname> <given-names>M</given-names></name> <etal/></person-group> <article-title>Induction of HIV neutralizing antibody lineages in mice with diverse precursor repertoires</article-title>. <source>Cell</source> (<year>2016</year>) <volume>166</volume>(<issue>6</issue>):<fpage>1471</fpage>&#x02013;<lpage>84.e18</lpage>.<pub-id pub-id-type="doi">10.1016/j.cell.2016.07.029</pub-id><pub-id pub-id-type="pmid">27610571</pub-id></citation></ref>
<ref id="B27"><label>27</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schramm</surname> <given-names>CA</given-names></name> <name><surname>Sheng</surname> <given-names>Z</given-names></name> <name><surname>Zhang</surname> <given-names>Z</given-names></name> <name><surname>Mascola</surname> <given-names>JR</given-names></name> <name><surname>Kwong</surname> <given-names>PD</given-names></name> <name><surname>Shapiro</surname> <given-names>L</given-names></name></person-group>. <article-title>SONAR: a high-throughput pipeline for inferring antibody ontogenies from longitudinal sequencing of B cell transcripts</article-title>. <source>Front Immunol</source> (<year>2016</year>) <volume>7</volume>:<fpage>372</fpage>.<pub-id pub-id-type="doi">10.3389/fimmu.2016.00372</pub-id><pub-id pub-id-type="pmid">27708645</pub-id></citation></ref>
<ref id="B28"><label>28</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lombana</surname> <given-names>TN</given-names></name> <name><surname>Dillon</surname> <given-names>M</given-names></name> <name><surname>Bevers</surname> <given-names>J</given-names> <suffix>III</suffix></name> <name><surname>Spiess</surname> <given-names>C</given-names></name></person-group>. <article-title>Optimizing antibody expression by using the naturally occurring framework diversity in a live bacterial antibody display system</article-title>. <source>Sci Rep</source> (<year>2015</year>) <volume>5</volume>:<fpage>17488</fpage>.<pub-id pub-id-type="doi">10.1038/srep17488</pub-id></citation></ref>
<ref id="B29"><label>29</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Klein</surname> <given-names>F</given-names></name> <name><surname>Diskin</surname> <given-names>R</given-names></name> <name><surname>Scheid</surname> <given-names>JF</given-names></name> <name><surname>Gaebler</surname> <given-names>C</given-names></name> <name><surname>Mouquet</surname> <given-names>H</given-names></name> <name><surname>Georgiev</surname> <given-names>IS</given-names></name> <etal/></person-group> <article-title>Somatic mutations of the immunoglobulin framework are generally required for broad and potent HIV-1 neutralization</article-title>. <source>Cell</source> (<year>2013</year>) <volume>153</volume>(<issue>1</issue>):<fpage>126</fpage>&#x02013;<lpage>38</lpage>.<pub-id pub-id-type="doi">10.1016/j.cell.2013.03.018</pub-id><pub-id pub-id-type="pmid">23540694</pub-id></citation></ref>
<ref id="B30"><label>30</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>T</given-names></name> <name><surname>Georgiev</surname> <given-names>I</given-names></name> <name><surname>Wu</surname> <given-names>X</given-names></name> <name><surname>Yang</surname> <given-names>ZY</given-names></name> <name><surname>Dai</surname> <given-names>K</given-names></name> <name><surname>Finzi</surname> <given-names>A</given-names></name> <etal/></person-group> <article-title>Structural basis for broad and potent neutralization of HIV-1 by antibody VRC01</article-title>. <source>Science</source> (<year>2010</year>) <volume>329</volume>(<issue>5993</issue>):<fpage>811</fpage>&#x02013;<lpage>7</lpage>.<pub-id pub-id-type="doi">10.1126/science.1192819</pub-id><pub-id pub-id-type="pmid">20616231</pub-id></citation></ref>
<ref id="B31"><label>31</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koenig</surname> <given-names>P</given-names></name> <name><surname>Lee</surname> <given-names>CV</given-names></name> <name><surname>Walters</surname> <given-names>BT</given-names></name> <name><surname>Janakiraman</surname> <given-names>V</given-names></name> <name><surname>Stinson</surname> <given-names>J</given-names></name> <name><surname>Patapoff</surname> <given-names>TW</given-names></name> <etal/></person-group> <article-title>Mutational landscape of antibody variable domains reveals a switch modulating the interdomain conformational dynamics and antigen binding</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2017</year>) <volume>114</volume>(<issue>4</issue>):<fpage>E486</fpage>&#x02013;<lpage>95</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.1613231114</pub-id><pub-id pub-id-type="pmid">28057863</pub-id></citation></ref>
<ref id="B32"><label>32</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ohm-Laursen</surname> <given-names>L</given-names></name> <name><surname>Nielsen</surname> <given-names>M</given-names></name> <name><surname>Larsen</surname> <given-names>SR</given-names></name> <name><surname>Barington</surname> <given-names>T</given-names></name></person-group>. <article-title>No evidence for the use of DIR, D-D fusions, chromosome 15 open reading frames or VH replacement in the peripheral repertoire was found on application of an improved algorithm, JointML, to 6329 human immunoglobulin H rearrangements</article-title>. <source>Immunology</source> (<year>2006</year>) <volume>119</volume>(<issue>2</issue>):<fpage>265</fpage>&#x02013;<lpage>77</lpage>.<pub-id pub-id-type="doi">10.1111/j.1365-2567.2006.02431.x</pub-id><pub-id pub-id-type="pmid">17005006</pub-id></citation></ref>
<ref id="B33"><label>33</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dorner</surname> <given-names>T</given-names></name> <name><surname>Brezinschek</surname> <given-names>HP</given-names></name> <name><surname>Brezinschek</surname> <given-names>RI</given-names></name> <name><surname>Foster</surname> <given-names>SJ</given-names></name> <name><surname>DomiatiSaad</surname> <given-names>R</given-names></name> <name><surname>Lipsky</surname> <given-names>PE</given-names></name></person-group>. <article-title>Analysis of the frequency and pattern of somatic mutations within nonproductively rearranged human variable heavy chain genes</article-title>. <source>J Immunol</source> (<year>1997</year>) <volume>158</volume>(<issue>6</issue>):<fpage>2779</fpage>&#x02013;<lpage>89</lpage>.</citation></ref>
<ref id="B34"><label>34</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McCoy</surname> <given-names>CO</given-names></name> <name><surname>Bedford</surname> <given-names>T</given-names></name> <name><surname>Minin</surname> <given-names>VN</given-names></name> <name><surname>Bradley</surname> <given-names>P</given-names></name> <name><surname>Robins</surname> <given-names>H</given-names></name> <name><surname>Matsen</surname> <given-names>FA</given-names> <suffix>IV</suffix></name></person-group>. <article-title>Quantifying evolutionary constraints on B-cell affinity maturation</article-title>. <source>Philos Trans R Soc Lond B Biol Sci</source> (<year>2015</year>) <volume>370</volume>(<issue>1676</issue>):<lpage>20140244</lpage>.<pub-id pub-id-type="doi">10.1098/rstb.2014.0244</pub-id><pub-id pub-id-type="pmid">26194758</pub-id></citation></ref>
<ref id="B35"><label>35</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blake</surname> <given-names>JD</given-names></name> <name><surname>Cohen</surname> <given-names>FE</given-names></name></person-group>. <article-title>Pairwise sequence alignment below the twilight zone</article-title>. <source>J Mol Biol</source> (<year>2001</year>) <volume>307</volume>(<issue>2</issue>):<fpage>721</fpage>&#x02013;<lpage>35</lpage>.<pub-id pub-id-type="doi">10.1006/jmbi.2001.4495</pub-id><pub-id pub-id-type="pmid">11254392</pub-id></citation></ref>
<ref id="B36"><label>36</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Murphy</surname> <given-names>K</given-names></name></person-group>. <source>Janeway&#x02019;s Immunobiology</source>. <publisher-loc>London, New York</publisher-loc>: <publisher-name>Garland Science, Taylor &#x00026; Francis Group, LLC.</publisher-name> (<year>2014</year>).</citation></ref>
<ref id="B37"><label>37</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Henry Dunand</surname> <given-names>CJ</given-names></name> <name><surname>Wilson</surname> <given-names>PC</given-names></name></person-group>. <article-title>Restricted, canonical, stereotyped and convergent immunoglobulin responses</article-title>. <source>Philos Trans R Soc Lond B Biol Sci</source> (<year>2015</year>) <volume>370</volume>(<issue>1676</issue>):<lpage>20140238</lpage>.<pub-id pub-id-type="doi">10.1098/rstb.2014.0238</pub-id><pub-id pub-id-type="pmid">26194752</pub-id></citation></ref>
<ref id="B38"><label>38</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Briney</surname> <given-names>BS</given-names></name> <name><surname>Willis</surname> <given-names>JR</given-names></name> <name><surname>McKinney</surname> <given-names>BA</given-names></name> <name><surname>Crowe</surname> <given-names>JE</given-names> <suffix>Jr</suffix></name></person-group>. <article-title>High-throughput antibody sequencing reveals genetic evidence of global regulation of the naive and memory repertoires that extends across individuals</article-title>. <source>Genes Immun</source> (<year>2012</year>) <volume>13</volume>(<issue>6</issue>):<fpage>469</fpage>&#x02013;<lpage>73</lpage>.<pub-id pub-id-type="doi">10.1038/gene.2012.20</pub-id></citation></ref>
<ref id="B39"><label>39</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Briney</surname> <given-names>BS</given-names></name> <name><surname>Willis</surname> <given-names>JR</given-names></name> <name><surname>Crowe</surname> <given-names>JE</given-names> <suffix>Jr</suffix></name></person-group>. <article-title>Human peripheral blood antibodies with long HCDR3s are established primarily at original recombination using a limited subset of germline genes</article-title>. <source>PLoS One</source> (<year>2012</year>) <volume>7</volume>(<issue>5</issue>):<fpage>e36750</fpage>.<pub-id pub-id-type="doi">10.1371/journal.pone.0036750</pub-id><pub-id pub-id-type="pmid">22590602</pub-id></citation></ref>
<ref id="B40"><label>40</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wagner</surname> <given-names>SD</given-names></name> <name><surname>Milstein</surname> <given-names>C</given-names></name> <name><surname>Neuberger</surname> <given-names>MS</given-names></name></person-group>. <article-title>Codon bias targets mutation</article-title>. <source>Nature</source> (<year>1995</year>) <volume>376</volume>(<issue>6543</issue>):<fpage>732</fpage>.<pub-id pub-id-type="doi">10.1038/376732a0</pub-id></citation></ref>
<ref id="B41"><label>41</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Laehnemann</surname> <given-names>D</given-names></name> <name><surname>Borkhardt</surname> <given-names>A</given-names></name> <name><surname>McHardy</surname> <given-names>AC</given-names></name></person-group>. <article-title>Denoising DNA deep sequencing data-high-throughput sequencing errors and their correction</article-title>. <source>Brief Bioinform</source> (<year>2016</year>) <volume>17</volume>(<issue>1</issue>):<fpage>154</fpage>&#x02013;<lpage>79</lpage>.<pub-id pub-id-type="doi">10.1093/bib/bbv029</pub-id><pub-id pub-id-type="pmid">26026159</pub-id></citation></ref>
<ref id="B42"><label>42</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Venturi</surname> <given-names>V</given-names></name> <name><surname>Quigley</surname> <given-names>MF</given-names></name> <name><surname>Greenaway</surname> <given-names>HY</given-names></name> <name><surname>Ng</surname> <given-names>PC</given-names></name> <name><surname>Ende</surname> <given-names>ZS</given-names></name> <name><surname>McIntosh</surname> <given-names>T</given-names></name> <etal/></person-group> <article-title>A mechanism for TCR sharing between T cell subsets and individuals revealed by pyrosequencing</article-title>. <source>J Immunol</source> (<year>2011</year>) <volume>186</volume>(<issue>7</issue>):<fpage>4285</fpage>&#x02013;<lpage>94</lpage>.<pub-id pub-id-type="doi">10.4049/jimmunol.1003898</pub-id><pub-id pub-id-type="pmid">21383244</pub-id></citation></ref>
<ref id="B43"><label>43</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Doria-Rose</surname> <given-names>NA</given-names></name> <name><surname>Schramm</surname> <given-names>CA</given-names></name> <name><surname>Gorman</surname> <given-names>J</given-names></name> <name><surname>Moore</surname> <given-names>PL</given-names></name> <name><surname>Bhiman</surname> <given-names>JN</given-names></name> <name><surname>DeKosky</surname> <given-names>BJ</given-names></name> <etal/></person-group> <article-title>Developmental pathway for potent V1V2-directed HIV-neutralizing antibodies</article-title>. <source>Nature</source> (<year>2014</year>) <volume>509</volume>(<issue>7498</issue>):<fpage>55</fpage>&#x02013;<lpage>62</lpage>.<pub-id pub-id-type="doi">10.1038/nature13036</pub-id><pub-id pub-id-type="pmid">24590074</pub-id></citation></ref>
<ref id="B44"><label>44</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>X</given-names></name> <name><surname>Zhang</surname> <given-names>Z</given-names></name> <name><surname>Schramm</surname> <given-names>CA</given-names></name> <name><surname>Joyce</surname> <given-names>MG</given-names></name> <name><surname>Kwon</surname> <given-names>YD</given-names></name> <name><surname>Zhou</surname> <given-names>T</given-names></name> <etal/></person-group> <article-title>Maturation and diversity of the VRC01-antibody lineage over 15 years of chronic HIV-1 infection</article-title>. <source>Cell</source> (<year>2015</year>) <volume>161</volume>(<issue>3</issue>):<fpage>470</fpage>&#x02013;<lpage>85</lpage>.<pub-id pub-id-type="doi">10.1016/j.cell.2015.03.004</pub-id><pub-id pub-id-type="pmid">25865483</pub-id></citation></ref>
<ref id="B45"><label>45</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>X</given-names></name> <name><surname>Zhou</surname> <given-names>T</given-names></name> <name><surname>Zhu</surname> <given-names>J</given-names></name> <name><surname>Zhang</surname> <given-names>B</given-names></name> <name><surname>Georgiev</surname> <given-names>I</given-names></name> <name><surname>Wang</surname> <given-names>C</given-names></name> <etal/></person-group> <article-title>Focused evolution of HIV-1 neutralizing antibodies revealed by structures and deep sequencing</article-title>. <source>Science</source> (<year>2011</year>) <volume>333</volume>(<issue>6049</issue>):<fpage>1593</fpage>&#x02013;<lpage>602</lpage>.<pub-id pub-id-type="doi">10.1126/science.1207532</pub-id><pub-id pub-id-type="pmid">21835983</pub-id></citation></ref>
<ref id="B46"><label>46</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bhiman</surname> <given-names>JN</given-names></name> <name><surname>Anthony</surname> <given-names>C</given-names></name> <name><surname>Doria-Rose</surname> <given-names>NA</given-names></name> <name><surname>Karimanzira</surname> <given-names>O</given-names></name> <name><surname>Schramm</surname> <given-names>CA</given-names></name> <name><surname>Khoza</surname> <given-names>T</given-names></name> <etal/></person-group> <article-title>Viral variants that initiate and drive maturation of V1V2-directed HIV-1 broadly neutralizing antibodies</article-title>. <source>Nat Med</source> (<year>2015</year>) <volume>21</volume>(<issue>11</issue>):<fpage>1332</fpage>&#x02013;<lpage>6</lpage>.<pub-id pub-id-type="doi">10.1038/nm.3963</pub-id><pub-id pub-id-type="pmid">26457756</pub-id></citation></ref>
<ref id="B47"><label>47</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edgar</surname> <given-names>RC</given-names></name></person-group>. <article-title>Search and clustering orders of magnitude faster than BLAST</article-title>. <source>Bioinformatics</source> (<year>2010</year>) <volume>26</volume>(<issue>19</issue>):<fpage>2460</fpage>&#x02013;<lpage>1</lpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/btq461</pub-id><pub-id pub-id-type="pmid">20709691</pub-id></citation></ref>
<ref id="B48"><label>48</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Crooks</surname> <given-names>GE</given-names></name> <name><surname>Hon</surname> <given-names>G</given-names></name> <name><surname>Chandonia</surname> <given-names>JM</given-names></name> <name><surname>Brenner</surname> <given-names>SE</given-names></name></person-group>. <article-title>WebLogo: a sequence logo generator</article-title>. <source>Genome Res</source> (<year>2004</year>) <volume>14</volume>(<issue>6</issue>):<fpage>1188</fpage>&#x02013;<lpage>90</lpage>.<pub-id pub-id-type="doi">10.1101/gr.849004</pub-id><pub-id pub-id-type="pmid">15173120</pub-id></citation></ref>
</ref-list>
<fn-group>
<fn id="fn1"><p><sup>1</sup><uri xlink:href="https://github.com/scharch/SONAR/tree/master/sonar/mGSSP">https://github.com/scharch/SONAR/tree/master/sonar/mGSSP</uri>.</p></fn>
<fn id="fn2"><p><sup>2</sup><uri xlink:href="https://github.com/scharch/SONAR">https://github.com/scharch/SONAR</uri>.</p></fn>
<fn id="fn3"><p><sup>3</sup><uri xlink:href="http://cgi.cse.unsw.edu.au/&#x0007E;ihmmune/IgPdb/index.php">http://cgi.cse.unsw.edu.au/&#x0007E;ihmmune/IgPdb/index.php</uri>.</p></fn>
</fn-group>
</back>
</article>