<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2018.00033</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Immune Repertoire Sequencing Using Molecular Identifiers Enables Accurate Clonality Discovery and Clone Size Quantification</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Ma</surname> <given-names>Ke-Yue</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/433963"/>
</contrib>
<contrib contrib-type="author">
<name><surname>He</surname> <given-names>Chenfeng</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/455086"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wendel</surname> <given-names>Ben S.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Williams</surname> <given-names>Chad M.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/436864"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Xiao</surname> <given-names>Jun</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Yang</surname> <given-names>Hui</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://frontiersin.org/people/u/479310"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Jiang</surname> <given-names>Ning</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="cor1">&#x0002A;</xref>
<uri xlink:href="http://frontiersin.org/people/u/358571"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Institute for Cellular and Molecular Biology, The University of Texas at Austin</institution>, <addr-line>Austin, TX</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Biomedical Engineering, Cockrell School of Engineering, The University of Texas at Austin</institution>, <addr-line>Austin, TX</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>McKetta Department of Chemical Engineering, Cockrell School of Engineering, The University of Texas at Austin</institution>, <addr-line>Austin, TX</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>ImmuDX, LLC</institution>, <addr-line>Austin, TX</addr-line>, <country>United States</country></aff>
<aff id="aff5"><sup>5</sup><institution>School of Life Sciences, Northwestern Polytechnical University</institution>, <addr-line>Xi&#x02019;an, Shaanxi</addr-line>, <country>China</country></aff>
<aff id="aff6"><sup>6</sup><institution>Research Center of Special Environmental Biomechanics &#x00026; Medical Engineering</institution>, <addr-line>Xi&#x02019;an, Shaanxi</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Gur Yaari, Bar-Ilan University, Israel</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Christopher Vollmers, University of California, Santa Cruz, United States; Mikhail Shugay, Institute of Bioorganic Chemistry (RAS), Russia</p></fn>
<corresp content-type="corresp" id="cor1">&#x0002A;Correspondence: Ning Jiang, <email>jiang&#x00040;austin.utexas.edu</email></corresp>
<fn fn-type="other" id="fn001"><p><sup>&#x02020;</sup>These authors have contributed equally to this work.</p></fn>
<fn fn-type="other" id="fn002"><p>Specialty section: This article was submitted to T Cell Biology, a section of the journal Frontiers in Immunology</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>05</day>
<month>02</month>
<year>2018</year>
</pub-date>
<pub-date pub-type="collection">
<year>2018</year>
</pub-date>
<volume>9</volume>
<elocation-id>33</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>09</month>
<year>2017</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>01</month>
<year>2018</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2018 Ma, He, Wendel, Williams, Xiao, Yang and Jiang.</copyright-statement>
<copyright-year>2018</copyright-year>
<copyright-holder>Ma, He, Wendel, Williams, Xiao, Yang and Jiang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) or licensor are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Unique molecular identifiers (MIDs) have been demonstrated to effectively improve immune repertoire sequencing (IR-seq) accuracy, especially to identify somatic hypermutations in antibody repertoire sequencing. However, evaluating the sensitivity to detect rare T cells and the degree of clonal expansion in IR-seq has been difficult due to the lack of knowledge of T cell receptor (TCR) RNA molecule copy number and a generalized approach to estimate T cell clone size from TCR RNA molecule quantification. This limited the application of TCR repertoire sequencing (TCR-seq) in clinical settings, such as detecting minimal residual disease in lymphoid malignancies after treatment, evaluating effectiveness of vaccination and assessing degree of infection. Here, we describe using an MID Clustering-based IR-Seq (MIDCIRS) method to quantitatively study TCR RNA molecule copy number and clonality in T cells. First, we demonstrated the necessity of performing MID sub-clustering to eliminate erroneous sequences. Further, we showed that MIDCIRS enables a sensitive detection of a single cell in as many as one million na&#x000EF;ve T cells and an accurate estimation of the degree of T cell clonal expression. The demonstrated accuracy, sensitivity, and wide dynamic range of MIDCIRS TCR-seq provide foundations for future applications in both basic research and clinical settings.</p>
</abstract>
<kwd-group>
<kwd>MID clustering-based IR-Seq TCR repertoire sequencing</kwd>
<kwd>molecular identifiers</kwd>
<kwd>sub-clustering</kwd>
<kwd>na&#x000EF;ve T cells</kwd>
<kwd>CMV-specific T cells</kwd>
</kwd-group>
<contract-num rid="cn01">R00AG040149, S10OD020072</contract-num>
<contract-num rid="cn02">1653866</contract-num>
<contract-num rid="cn03">F1785</contract-num>
<contract-num rid="cn04">1147222, 11672246</contract-num>
<contract-sponsor id="cn01">National Institutes of Health<named-content content-type="fundref-id">10.13039/100000002</named-content></contract-sponsor>
<contract-sponsor id="cn02">National Science Foundation<named-content content-type="fundref-id">10.13039/100000001</named-content></contract-sponsor>
<contract-sponsor id="cn03">Welch Foundation<named-content content-type="fundref-id">10.13039/100000928</named-content></contract-sponsor>
<contract-sponsor id="cn04">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content></contract-sponsor>
<counts>
<fig-count count="4"/>
<table-count count="1"/>
<equation-count count="8"/>
<ref-count count="48"/>
<page-count count="11"/>
<word-count count="8015"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="introduction">
<title>Introduction</title>
<p>Immune repertoire sequencing (IR-seq) has become a useful tool to quantify the composition of B or T cell antigen receptor repertoires in basic research, such as vaccination (<xref ref-type="bibr" rid="B1">1</xref>&#x02013;<xref ref-type="bibr" rid="B3">3</xref>), immune repertoire development (<xref ref-type="bibr" rid="B4">4</xref>&#x02013;<xref ref-type="bibr" rid="B9">9</xref>), and lymphocyte lineage tracking (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B9">9</xref>), as well as in various clinical settings, such as minimal residual disease (MRD) monitoring (<xref ref-type="bibr" rid="B10">10</xref>), hematopoietic stem cell transplant recovery monitoring (<xref ref-type="bibr" rid="B11">11</xref>), and cancer patient prognosis (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B13">13</xref>). However, early IR-seq experiments suffered from high PCR and sequencing errors that limited their ability to perform accurate repertoire diversity and abundance quantification. This bottleneck also limits the sensitivity of many IR-seq-based assays, such as MRD monitoring. Recently, we and others introduced molecular identifiers (MIDs) to IR-seq and DNA/RNA sequencing to reduce errors by tracking each RNA molecule through PCR and sequencing. This approach has significantly improved the accuracy of repertoire profiling (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B14">14</xref>&#x02013;<xref ref-type="bibr" rid="B19">19</xref>), especially to distinguish antibody somatic hypermutations from PCR and sequencing errors. However, several challenges remain regarding how to use MIDs correctly and how to use MIDs for cell clone size estimate. First, erroneous MIDs resulting from PCR or sequencing errors make accurate MID counting difficult. Second, there is a lack of general guidelines of required sequencing depth to saturate MID counts. Third, how to use RNA molecular counting to estimate T cell clone size has yet to be established.</p>
<p>These challenges become roadblocks to accurately quantify T cell receptor (TCR) or BCR RNA molecule copy number, which is important in estimating clonal expansion and identifying rare clones. Robins et al. developed QuanTILfy to attempt to address this problem by counting TILs and assessing T cell clonality in tissue samples through droplet digital PCR (dPCR) of rearranged TCR&#x003B2; loci (<xref ref-type="bibr" rid="B20">20</xref>). However, by partitioning TCR V&#x003B2; into eight non-overlapping subgroups, this method lacks the sensitivity to identify unique CDR3 of each clonality, not to mention rare clones. Therefore, a more comprehensive method to quantify TCR or antibody transcripts with high sensitivity while retaining accurate clonal diversity is needed for both standardizing basic IR-seq studies and applying it in clinical decision-making, such as detecting MRD in lymphoid malignancies after treatment, evaluating effectiveness of vaccination, and assessing degree of infection.</p>
<p>We recently developed a more generalized approach with reduced MID length to identify each individual RNA molecule using a sequence-similarity-based clustering method to separate sequencing reads into sub-clusters within a group of sequencing reads that have the same MID. We applied this MID Clustering-based IR-Seq (MIDCIRS) to study age-related antibody repertoire development and diversification during acute malaria (<xref ref-type="bibr" rid="B9">9</xref>). In this study, we applied MIDCIRS to TCR [MIDCIRS TCR repertoire sequencing (TCR-seq)] and used CD8<sup>&#x0002B;</sup> T cells as a test bed to build a model to count TCR RNA molecule copy number based on input cell numbers, percentage of RNA input, and sequencing depth. We also demonstrated a significant improvement in detection sensitivity. A previous study using a different repertoire sequencing methodology reported the capacity to resolve one in 10,000 cells (<xref ref-type="bibr" rid="B21">21</xref>). With MIDCIRS TCR-seq, we were able to detect one unique T cell clone in 1,000,000 T cells. In addition, we applied MIDCIRS TCR-seq to examine T cell clonal expansion in CMV infection and showed that sensitive and accurate quantification of the TCR RNA molecule copy number is essential to quantify a single-cell&#x02019;s worth of TCR transcripts and to assess the degree of clonal expansion. In summary, we showed the significance of the sub-clustering step of MIDCIRS in preventing false MID group generation, which enabled highly accurate clonal type discovery. This study provides a framework for leveraging the sensitivity and accuracy of molecular barcoded IR-seq in MRD detection and assessing clonal expansion in infection and vaccination.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S2-1">
<title>Na&#x000EF;ve CD8<sup>&#x0002B;</sup> T Cell Sorting</title>
<p>Human leukocyte reduction system chambers were obtained from de-identified donors at We Are Blood (Austin, TX, USA) with strict adherence to guidelines from the Institutional Review Board of the University of Texas at Austin. CD8<sup>&#x0002B;</sup> T cell enrichment was done following the protocol described previously (<xref ref-type="bibr" rid="B22">22</xref>) using RosetteSep CD8<sup>&#x0002B;</sup> T Cell Enrichment Cocktail (STEMCELL) together with Ficoll-Paque (GE Healthcare). Then, RBCs were lysed using ACK Lysing Buffer (Lonza). After washing in phosphate-buffered saline with fetal bovine serum, the cell mixture was passed through a cell strainer (Corning) and ready for use. Na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells were FACS-sorted into RLT Plus buffer (Qiagen) supplemented with 1% &#x003B2;-mercaptoethanol (Sigma) based on the phenotype of CD8<sup>&#x0002B;</sup>CD4<sup>-</sup>CCR7<sup>&#x0002B;</sup>CD45RA<sup>&#x0002B;</sup> using BD FACSAria II cell sorter.</p>
</sec>
<sec id="S2-2">
<title>CMV CD8<sup>&#x0002B;</sup> T Cell Enrichment and Sorting</title>
<p>CMVpp65:482-490 (NLVPMVATV) was used to prepare streptamers as previously described (<xref ref-type="bibr" rid="B23">23</xref>). Miltenyi anti-phycoerythrin microbeads and magnetic column were used to bind and enrich CMVpp65-specific T cells (<xref ref-type="bibr" rid="B22">22</xref>). The flow-through was collected for background staining. The enriched fraction was eluted off the column and washed into cell buffer. The following antibody panel was used to stain both the enriched and flow-through fractions: CD4, CD14, CD16, CD19, CD32, and CD56 (BioLegend) as a dump channel to stain residual non-CD8 T cells, and CD45RA, CCR7, CD27, and IL7R (BioLegend). 7-aminoactinomycin D was used as a viability marker. Dump<sup>&#x02212;</sup>Streptmer<sup>&#x0002B;</sup>CD45RA<sup>&#x0002B;</sup>CCR7<sup>&#x02212;</sup>CD27<sup>&#x02212;</sup>IL7R<sup>lo</sup> live T cells were sorted into RLT Plus buffer supplemented with 1% &#x003B2;-mercaptoethanol using BD FACSAria II cell sorter.</p>
</sec>
<sec id="S2-3">
<title>Bulk TCR Library Generation and Sequencing</title>
<p>Total RNA was purified using All Prep DNA/RNA kit (Qiagen) following the manufacturer&#x02019;s protocol. Library preparation and QC were similar to protocols described previously (<xref ref-type="bibr" rid="B9">9</xref>) using TCR primers (Table S5 in Supplementary Material). Reads of the same library from all runs were combined and analyzed.</p>
</sec>
<sec id="S2-4">
<title>dPCR of TCR</title>
<p>Total RNA purified from sorted CD8<sup>&#x0002B;</sup> T cells and cultured CMV-specific CD8<sup>&#x0002B;</sup> T cell lines were reverse transcribed with polyT primers (Table S5 in Supplementary Material) using Superscript III in 20&#x02009;&#x000B5;l reaction following the manufacturer&#x02019;s protocol. 2&#x02009;&#x000B5;l of cDNA was subsequently used on QuantStudio 3D dPCR system following manufacturer&#x02019;s protocol.</p>
</sec>
<sec id="S2-5">
<title>Preliminary Read Processing</title>
<p>We followed the similar procedure as described previously to generate consensus sequences (<xref ref-type="bibr" rid="B9">9</xref>). First, only reads that have exact TCR constant sequences were kept for further analysis. These reads were then cut to 150&#x02009;nt starting from constant region to eliminate high error-prone region at the end of reads. These preprocessed reads were split into MID groups according to 12-nt barcodes.</p>
</sec>
<sec id="S2-6">
<title>MID Sub-Cluster Generating and Filtering</title>
<p>For each MID group, a quality threshold clustering was used to group reads derived from a common ancestor RNA molecule and separate reads derived from distinct RNAs as previously described (<xref ref-type="bibr" rid="B9">9</xref>). Briefly, a Levenshtein distance of 15% of the read length was used as the threshold (<xref ref-type="bibr" rid="B9">9</xref>). For each subgroup, a consensus sequence was built based on the average nucleotide at each position, weighted by the quality score. In the case that there were only two reads in an MID subgroup, we only considered them useful reads if both were identical. Each MID subgroup is equivalent to an RNA molecule. Next, we merged all of the identical consensus to form unique consensus sequences. Further, we applied filtering of unique consensus sequences after sub-cluster generation by (a) removing non-functional TCR sequences and (b) removing sequences with lower MID counts that are one Levenshtein distance away from the other. Then, for each unique consensus sequence, we removed MID sub-clusters if their reads are less than 20% of maximum read count based on the fitting of two negative binomial distribution (Figure S5 in Supplementary Material). Scripts for this section can be downloaded at <uri xlink:href="https://github.com/utjianglab/MIDCIRS">https://github.com/utjianglab/MIDCIRS</uri>.</p>
</sec>
<sec id="S2-7">
<title>Theoretical Percentage of MIDs That Need Sub-Clustering</title>
<p>We modeled the process of MID labeling as a Poisson distribution. Given the total number of MIDs being <italic>M</italic> and the number of target molecules being <italic>N</italic>, the probability that a unique MID will occur <italic>k</italic> time(s) is:
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mfrac><mml:mi>N</mml:mi><mml:mi>M</mml:mi></mml:mfrac><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>k</mml:mi></mml:msup></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>!</mml:mo></mml:mrow></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mi>N</mml:mi><mml:mi>M</mml:mi></mml:mfrac></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>Thus, <italic>P</italic><sub>0</sub> and <italic>P</italic><sub>1</sub> are the probability that a MID will be tagged 0 and 1 time, respectively, and the percentage of MIDs that need sub-clustering, <italic>F</italic>(<italic>k</italic>&#x02009;&#x0003E;&#x02009;1), is given by:
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mrow><mml:mi>F</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>&#x0003E;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mi>N</mml:mi><mml:mi>M</mml:mi></mml:mfrac></mml:mrow></mml:msup><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mi>N</mml:mi><mml:mi>M</mml:mi></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mi>N</mml:mi><mml:mi>M</mml:mi></mml:mfrac></mml:mrow></mml:msup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mi>N</mml:mi><mml:mi>M</mml:mi></mml:mfrac></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>With over 16 million MID combinations from 12 random nucleotides, when the number of target molecules, <italic>N</italic> is less than 5,000,000, Eq.&#x02009;<xref ref-type="disp-formula" rid="E2">2</xref> is an approximate linear function (Figure <xref ref-type="fig" rid="F1">1</xref>B).</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>MID Clustering-based IR-Seq improves accuracy of T cell receptor (TCR) diversity estimation with sub-clustering. <bold>(A)</bold> The percentage of observed molecular identifiers (MIDs) containing sub-clusters is linearly dependent on RNA input, which is defined as cell number multiplied by percentage of RNA (e.g., 20,000 cells with 10%RNA is equivalent to 2,000 RNA input). Line represents linear regression fit, <italic>F</italic>-test on the slope, <italic>p</italic>&#x02009;&#x0003C;&#x02009;10<sup>&#x02212;9</sup>. <bold>(B)</bold> The theoretical percentage of MIDs with sub-clusters is approximately linearly dependent on copies of target molecules when copies of target molecules are less than 5,000,000 (bottom right insert). The theoretical percentage of MIDs with sub-clusters was calculated by Eq.&#x02009;<xref ref-type="disp-formula" rid="E2">2</xref> in Section &#x0201C;<xref ref-type="sec" rid="S2">Materials and Methods</xref>.&#x0201D; <bold>(C)</bold> Rarefaction curve of unique complementarity-determining regions 3 (CDR3s) with or without sub-clustering. Number of unique CDR3s in three libraries made with three different RNA inputs from sorted one million na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells are shown here. Data from other cell inputs are in Figure S2 in Supplementary Material. <bold>(D)</bold> Illustration of consensus TCR sequence building without (top) and with (bottom) sub-clustering. Top: without sub-clustering, chimera sequences are generated when different TCR RNA molecules are tagged with the same MID; bottom: TCR RNA molecules that are tagged with same MID are sub-clustered to reveal truly represented TCR sequences. Short vertical black lines indicate nucleotide differences between two TCR sequences.</p></caption>
<graphic xlink:href="fimmu-09-00033-g001.tif"/>
</fig>
</sec>
<sec id="S2-8">
<title>Diversity Coverage and RNA Copy Number Simulation</title>
<p>The estimation of diversity will be affected by the initial RNA input (percentage of initial RNA used to construct the sequencing library). We used a statistical model to estimate the diversity coverage for the na&#x000EF;ve T cells we sorted based on RNA sampling depth.</p>
<p>For <italic>N</italic> observed RNA molecules, there are <italic>K</italic> different RNA clones. The RNA molecule copy number of each clone is <italic>m</italic><sub>i</sub> (<italic>i</italic>&#x02208;(1,<italic>K</italic>)), whose sum equals <italic>N</italic>. After fitting the data, <italic>m</italic><sub>i</sub> follows a power law distribution (Figure S9 in Supplementary Material):
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mrow><mml:msub><mml:mi>m</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math></disp-formula>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:msubsup><mml:mi>x</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B1;</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x0003E;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:math></disp-formula>
where, <italic>m</italic> is the RNA molecule copy number per cell, which is a constant across all T cells (see Figure <xref ref-type="fig" rid="F3">3</xref>C). <italic>x</italic><sub>i</sub> represents the cell numbers of each clone, which follows a power law distribution (<xref ref-type="bibr" rid="B24">24</xref>), and the parameter &#x003B1; was fitted with an algorithm combining maximum-likelihood fitting and goodness-of-fit test based on Kolmogorov&#x02013;Smirnov statistic (<xref ref-type="bibr" rid="B25">25</xref>) &#x0201C;fit_power_law&#x0201D; function in R package igraph was applied (<xref ref-type="bibr" rid="B26">26</xref>).</p>
<p>Specifically, we fitted the RNA molecule distribution (Figure S9 in Supplementary Material) with Eq.&#x02009;<xref ref-type="disp-formula" rid="E5">5</xref>:
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>m</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mtext>min</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mi>m</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>m</mml:mi><mml:mrow><mml:mtext>min</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B1;</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x0003E;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>Since &#x0201C;<italic>m</italic>&#x0201D; is a constant (see Figure <xref ref-type="fig" rid="F3">3</xref>C), the alpha in Eqs&#x02009;<xref ref-type="disp-formula" rid="E4">4</xref> and <xref ref-type="disp-formula" rid="E5">5</xref> should be equal. We fitted across all libraries on log&#x02013;log scale, and the average slope was taken as &#x003B1; in the above model.</p>
<p>When we sample <italic>n</italic> RNA molecules from this population, the expected detected diversity, <italic>E</italic>(D), can be calculated as the following:
<disp-formula id="E6"><label>(6)</label><mml:math id="M6"><mml:mrow><mml:mi>E</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>D</mml:mi><mml:mo>&#x0007C;</mml:mo><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>K</mml:mi></mml:msubsup></mml:mstyle><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd columnalign="center"><mml:mi>n</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mi>N</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>n</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>K</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>And <italic>x</italic><sub>i</sub> can be sampled from the fitted power law distribution.</p>
<p>Then, the percentage of the RNA diversity coverage, <italic>P</italic>(D), can be estimated as:
<disp-formula id="E7"><label>(7)</label><mml:math id="M7"><mml:mrow><mml:mi>P</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>D</mml:mi><mml:mo>&#x0007C;</mml:mo><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>E</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>D</mml:mi><mml:mo>&#x0007C;</mml:mo><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mi>K</mml:mi></mml:mfrac><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>We scaled the diversity coverage of unique CDR3s to the estimated diversity coverage with 90% RNA input, <italic>D</italic><sub>obs</sub>. We then used Eq.&#x02009;<xref ref-type="disp-formula" rid="E8">8</xref> to get estimated <italic>m</italic>:
<disp-formula id="E8"><label>(8)</label><mml:math id="M8"><mml:mrow><mml:munder><mml:mrow><mml:mtext>min</mml:mtext></mml:mrow><mml:mi>m</mml:mi></mml:munder><mml:mstyle displaystyle='true'><mml:munder><mml:mo>&#x02211;</mml:mo><mml:mi>i</mml:mi></mml:munder></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x0007C;</mml:mo><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mtext>obs</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>,</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p>
</sec>
<sec id="S2-9">
<title>Statistical Analysis</title>
<p>Mann&#x02013;Whitney <italic>U</italic> test was used to calculate the significance of copy number difference between pairs in na&#x000EF;ve, effector, effector memory, and central memory CD8<sup>&#x0002B;</sup> T cells and <italic>p</italic> values was adjusted with Benjamini&#x02013;Hochberg procedure. Adjusted <italic>p</italic>-value that was less than 0.05 was considered significant.</p>
</sec>
</sec>
<sec id="S3">
<title>Results</title>
<sec id="S3-1">
<title>MIDCIRS Sub-Clustering Improves Repertoire Diversity Estimation Accuracy</title>
<p>Molecular identifiers have been adopted in IR-seq and DNA/RNA sequencing to reduce error rate. However, during reverse transcription, multiple transcripts could stochastically be tagged with same MID. Previous strategies relied on increasing the length of MID to reduce the probability of non-unique MID tagging when the total RNA molecule copy number was either unknown or very large (<xref ref-type="bibr" rid="B27">27</xref>). However, longer MID length could reduce the efficiency of reverse transcription (<xref ref-type="bibr" rid="B28">28</xref>, <xref ref-type="bibr" rid="B29">29</xref>). Thus, we developed a more generalized approach (MIDCIRS) with reduced MID length. A sequence-similarity-based clustering method was implemented in MIDCIRS to separate sequencing reads into sub-clusters within a group of sequencing reads that have the same MID (<xref ref-type="bibr" rid="B9">9</xref>). Here, we developed metrics to validate the accuracy of this sub-clustering method. In addition, we demonstrated the robust ability of MIDCIRS to faithfully represent the diversity and abundance of the TCR repertoire using a large range of RNA inputs.</p>
<p>We reasoned that in order to comprehensively quantify the overall diversity, a large portion of its RNA must be sampled. However, this will inevitably increase the number of TCR transcripts that need to be tagged with MIDs, which increases the portion of MIDs tagging multiple TCR transcripts. We sought to closely examine the relationship between RNA input and multiple TCR RNA tagging by the same MID. The process of MID labeling can be modeled as a Poisson distribution (see <xref ref-type="sec" rid="S2">Materials and Methods</xref>). The percentage of MIDs with sub-clusters follows an approximate linear trend when the copies of target RNA molecules are less than 5,000,000 (Figure <xref ref-type="fig" rid="F1">1</xref>B). To experimentally validate this, we applied MIDCIRS TCR-seq on a range of sorted na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells (from 20,000 to 1 million) with three different RNA inputs (10, 30, and 50%) (Table S1 in Supplementary Material). We have previously used control template sequences and evaluated the clustering threshold that would separate TCR RNA molecules accidentally tagged with the same MID, which is 15% of the sequence length (<xref ref-type="bibr" rid="B9">9</xref>). As expected, we found that the observed percentage of MIDs that need sub-clustering is approximately linear with respect to copies of target RNA molecules used in this study (Figure <xref ref-type="fig" rid="F1">1</xref>A). With the highest amount of RNA molecules used in this study, approximately 8.5% of MIDs require further clustering, while previous method treated these sequences as ambiguous (<xref ref-type="bibr" rid="B17">17</xref>). Thus, MIDCIRS sub-clustering significantly improves repertoire diversity coverage.</p>
<p>To evaluate the accuracy of the sub-clustering step by an alternative means, we examined the TCR sequence lengths within MIDs that contain sub-clusters. We reasoned that if indeed each TCR RNA molecule was tagged with a unique MID, then the lengths of CDR3 for all reads would be identical under each MID. However, we showed that of the 8.5% of MIDs that contain sub-clusters, about 87% of MIDs contain TCR sequencing reads of different CDR3 lengths while only 13% have the same length for one million na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells (50% RNA input). After performing sub-clustering, over 97% of sub-clusters have a uniform length (Figure S1 in Supplementary Material), demonstrating the accuracy of sub-clustering step in MIDCIRS.</p>
<p>More importantly, to our surprise, we found that, without performing sub-clustering, the number of unique consensus sequences (unique CDR3 sequences) was overestimated, especially in samples with one million cells (Figure <xref ref-type="fig" rid="F1">1</xref>C; Figure S2 in Supplementary Material). This is because chimera sequences were generated in the consensus building step for two scenarios. In one scenario, multiple true TCR sequences could be tagged with the same MID and quality score weighted consensus building will generate chimera sequences (Figure <xref ref-type="fig" rid="F1">1</xref>D; Figure S3A in Supplementary Material). In the second scenario, PCR or sequencing errors on MIDs group multiple singletons (MIDs that contain only one read) under the new MID. If sub-clustering is applied, then these singletons will be separated and discarded under the singleton category. However, without sub-clustering, these singletons will be forced to generate a chimera sequence (Figure S3B in Supplementary Material). Taking together, these chimera sequences cause overestimation of the total TCR diversity. The percentage of chimera sequences can be as high as 47% (Table S1 in Supplementary Material). Thus, compared with previous IR-seq with MID method (<xref ref-type="bibr" rid="B17">17</xref>), MIDCIRS not only can increase diversity coverage of CDR3 but improve the accuracy of diversity estimation.</p>
</sec>
<sec id="S3-2">
<title>MID Read-Distribution-Based Barcode Correction Improves Accuracy and Sensitivity of Counting TCR Transcripts</title>
<p>Besides correcting PCR and sequencing errors, MIDs have also been used for absolute quantification of RNA molecule copy number in single-cell studies to improve precision (<xref ref-type="bibr" rid="B30">30</xref>&#x02013;<xref ref-type="bibr" rid="B33">33</xref>). Here, we demonstrated how to use MIDCIRS TCR-seq to digitally count TCR transcripts. The absolute quantification of TCR transcripts is fundamental for accurate clonal size estimation. We noticed that PCR and sequencing errors also affected MIDs, as seen in single-cell RNA sequencing studies (<xref ref-type="bibr" rid="B29">29</xref>, <xref ref-type="bibr" rid="B34">34</xref>), leading to an inflated number of RNA molecules when libraries were sequenced exhaustively with respective to the total TCR transcripts in the sample (Figure <xref ref-type="fig" rid="F2">2</xref>A; Figure S4 in Supplementary Material). To correct MID errors, we first removed singleton reads, which cannot be confidently used in generating MID groups due to sequencing errors. Then, we adopted a similar approach applied in single-cell RNA-seq by fitting the distribution of reads under each MID subgroup into two negative binomial distributions (Figure S5 in Supplementary Material) (<xref ref-type="bibr" rid="B34">34</xref>). Erroneous MIDs generated due to PCR errors generally have distinctively lower read counts compared with true MIDs. These two negative binomial distributions distinctly separated true MIDs from erroneous MIDs. MIDs with low read counts were removed accordingly (see <xref ref-type="sec" rid="S2">Materials and Methods</xref>). After MID correction, number of RNA molecules saturated across libraries (Figure <xref ref-type="fig" rid="F2">2</xref>A; Figure S4 in Supplementary Material).</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>MID Clustering-based IR-Seq is capable of accurate digital counting of T cell receptor (TCR) RNA molecules. <bold>(A)</bold> Rarefaction curve of detected TCR RNA molecules before and after error correction on molecular identifiers (MIDs) in 20,000 na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells for three RNA input amounts. Data from other cell inputs are in Figure S4 in Supplementary Material. <bold>(B)</bold> Comparison of rarefaction curve of detected RNA molecules and unique complementarity-determining regions 3 (CDR3s) in 20,000 na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells for three RNA input amounts. <bold>(C)</bold> Rarefaction curve of number of unique CDR3s with single RNA copy in 20,000 na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells for three RNA input amounts. Sequencing reads were subsampled to different depth and unique CDR3s were tallied. Data from other cell inputs are in Figure S6A in Supplementary Material. <bold>(D)</bold> The percentage of overlapping clones with single RNA copy at different sequencing depths by sub-sampling in 20,000 na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells for three RNA input amounts. The overlapping clones were compared between two adjacent sub-samplings and overlap percentage was calculated by dividing the number of overlapping clones by the total number of clones observed in the deeper sub-sampling. Data from other cell input are in Figure S6B in Supplementary Material.</p></caption>
<graphic xlink:href="fimmu-09-00033-g002.tif"/>
</fig>
<p>We found that a shallower sequencing depth is required to saturate unique CDR3s than RNA molecules (Figure <xref ref-type="fig" rid="F2">2</xref>B). In addition, the amount of diversity covered increased with increasing RNA input. Thus, to exhaustively measure the TCR repertoire diversity, with 30&#x02013;50% of RNA input, a sequencing depth equivalent to 10 times the cell number covers most of the CDR3 diversity (Figure <xref ref-type="fig" rid="F1">1</xref>C; Figure S2 in Supplementary Material), while a sequencing depth equivalent to about 100 times the relative RNA input (defined as cell number multiplied by percentage of RNA input) is required to saturate the RNA molecules (Figure <xref ref-type="fig" rid="F2">2</xref>A; Figure S4 in Supplementary Material). For example, 30% RNA of 20,000 cells is equivalent to 6,000 RNA input. Then, it takes about 600,000 reads to saturate the RNA molecules but only 200,000 reads to saturate the unique CDR3s (Figure <xref ref-type="fig" rid="F2">2</xref>A, middle panel).</p>
<p>After MID correction, with optimal sequencing depth, we stably detected TCR clones with a single TCR RNA molecule (single-copy clones with at least two identical sequencing reads). The number of single-copy clones saturates with adequate sequencing depth (Figure <xref ref-type="fig" rid="F2">2</xref>C; Figure S6A in Supplementary Material). Meanwhile, we compared the degree of overlapping clones within these single-copy clones at different sequencing depths. To do this, we subsampled each library to different fractions of the total reads. The overlapping clones were compared between two adjacent subsamples, and the overlap percentage was calculated by dividing the number of overlapping clones by the total number of clones observed in the deeper subsample. Thus, for total of 10 subsamples, 9 clonal overlap percentages were calculated and plotted with respect to sequencing depth (Figure <xref ref-type="fig" rid="F2">2</xref>D; Figure S6B in Supplementary Material). More than 90% of single-copy clones were repeatedly detected between the full sequencing reads and the 0.9 subsample fraction. The overlap percentage was above 80% for the latter part of curve (Figure <xref ref-type="fig" rid="F2">2</xref>D; Figure S6B in Supplementary Material), which suggested that we have reached optimal sequencing depth to detect single-copy TCR clones.</p>
</sec>
<sec id="S3-3">
<title>Estimating TCR RNA Molecule Copy Number and Validation with dPCR</title>
<p>From early analysis, we know that the diversity coverage of unique CDR3s increased as RNA input increased. Here, we performed an in depth analysis on the relationship between these two parameters and found that the diversity coverage of unique CDR3s increased significantly as the RNA input increased initially, then reached a plateau, which resulted in a nonlinear increasing of the diversity coverage of unique CDR3s (Figures <xref ref-type="fig" rid="F3">3</xref>A,B). We assumed that total diversity for a sample is the diversity discovered when combining all sequencing reads from 10, 30, and 50% RNA input libraries into a pseudo-90% RNA input. With 50% RNA, we could recover about 60% of total diversity (Figure <xref ref-type="fig" rid="F3">3</xref>B).</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>T cell receptor (TCR) RNA copy number per cell estimation and experimental validation. <bold>(A)</bold> Diversity coverage of unique productive complementarity-determining regions 3 with different RNA inputs and cell numbers (Line represents linear regression fit, <italic>F</italic>-test on the slope, <italic>R</italic><sup>2</sup>&#x02009;&#x0003E;&#x02009;0.99 and <italic>p</italic>&#x02009;&#x0003C;&#x02009;10<sup>&#x02212;3</sup> for all different RNA inputs). <bold>(B)</bold> Diversity coverages with different RNA inputs using 3 as a predicted TCR RNA molecule copy number per cell. Dashed line is the theoretical prediction (see <xref ref-type="sec" rid="S2">Materials and Methods</xref>); red dots are diversity coverages observed in libraries with different RNA inputs as illustrated in panel <bold>(A)</bold>, assuming diversity coverage at 90% RNA input is 1. <bold>(C)</bold> Digital PCR results of TCR RNA molecule copies per cell in different CD8<sup>&#x0002B;</sup> T cell subset (N, na&#x000EF;ve; CM, central memory; EM, effector memory; E, effector; NTC, no template control; n.s: <italic>p</italic>-value&#x02009;&#x0003E;&#x02009;0.05 by Mann&#x02013;Whitney <italic>U</italic> test).</p></caption>
<graphic xlink:href="fimmu-09-00033-g003.tif"/>
</fig>
<p>Since the observed diversity is dependent on total TCR RNA molecules in a sample, which is a function of TCR RNA molecule copy number per cell and RNA input percentage, we next sought to use a probability model to predict TCR RNA molecule copy number per cell using the observed diversity coverage of unique CDR3s as a function of RNA input percentage (see <xref ref-type="sec" rid="S2">Materials and Methods</xref>). We used the estimated diversity coverage of different RNA inputs, including 10, 30, and 50% RNA, as well as the computationally combined pseudo-40% (10&#x02009;&#x0002B;&#x02009;30%) and pseudo-90% RNA inputs as data points to fit the probability model. The best fit resulted in three copies of TCR RNA molecule per cell (Figure <xref ref-type="fig" rid="F3">3</xref>B). In another independent experiment, RNA from 20,000 and 100,000 na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells were evenly separated into five aliquots, respectively. Four of five aliquots were sequenced (Table S2 in Supplementary Material). Results showed that CDR3 diversity detected by MIDCIRS is very reproducible among the four aliquots and is also proportional to the cell input numbers. In addition, we bioinformatically combined the aliquots into pseudo-40, -60, and -80% of RNA inputs and fitted the diversity coverage using the probability model described in the Section &#x0201C;<xref ref-type="sec" rid="S2">Materials and Methods</xref>.&#x0201D; As with previously, the best fit resulted in three copies of TCR RNA molecule per cell (Figure S7 in Supplementary Material).</p>
<p>However, in order to apply this TCR RNA molecule copy number in estimating T cell clone size, we need to validate it using a different method and also test to see if different phenotypes of T cells might have different TCR RNA molecule copy numbers, which would be similar to the differences seeing in na&#x000EF;ve B cells and plasmablasts (<xref ref-type="bibr" rid="B35">35</xref>). Next, we validated TCR RNA molecule copy number using dPCR and found that various types of T cells have similar TCR RNA copies (8&#x02013;12 copies per cell) (Figure <xref ref-type="fig" rid="F3">3</xref>C). Thus, with MIDCIRS TCR-seq, we could achieve about 30% efficiency in recovering the target TCR RNA molecules, which is expected given dPCR in a nanoliter volume is more efficient than bulk PCR in tubes (<xref ref-type="bibr" rid="B36">36</xref>). This ratio also establishes a reference point for rare T cell clone frequency estimate using MIDCIRS method.</p>
</sec>
<sec id="S3-4">
<title>Detecting Single-Cell Worth of TCR RNA Using MIDCIRS</title>
<p>The lack of accurate and absolute quantitation of TCR clones limited the evaluation of the sensitivity of various IR-seq methods (<xref ref-type="bibr" rid="B37">37</xref>), which slowed the application of detecting rare TCR clones in both basic research and clinical practice. To address the detection sensitivity using MIDCIRS, we spiked-in control TCR RNA with varying copy numbers into na&#x000EF;ve T cells and validated the robustness of detecting spiked-in TCRs. 5, 20, and 5 copies of three spike-in cell lines with known TCR sequences were added into 20,000 and 100,000 na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells. 3, 13, and 3 copies of three spike-ins were reliably detected, respectively (Figure <xref ref-type="fig" rid="F4">4</xref>A).</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>MID Clustering-based IR-Seq is sensitive to detect both low copy and highly clonal expanded T cell receptors (TCRs). <bold>(A)</bold> Number of RNA molecules detected by sequencing for each spike-in TCR control sequences (the numbers in the legend denote copies of each TCR spike-in control sequence added). <bold>(B)</bold> Comparison of clone size distribution in na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells and CMVpp65-specific effector CD8<sup>&#x0002B;</sup> T cells (dashed line indicates TCR sequences with 20 copies of RNA molecules). <bold>(C)</bold> The percentage of RNA molecules that varying degree of clonally expanded complementarity-determining region 3 account for.</p></caption>
<graphic xlink:href="fimmu-09-00033-g004.tif"/>
</fig>
<p>We also analyzed the ability to detect a single T cell&#x02019;s worth of control RNA in a larger number of other T cells. We digitally counted the concentration of TCR RNA molecule from the Jurkat cell line and spiked-in 10 copies of TCR RNA into 20,000&#x02013;1,000,000 na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells (Table S1 in Supplementary Material). In all 1,000,000 cells we sequenced, we were capable of detecting Jurkat TCR sequences (Table <xref ref-type="table" rid="T1">1</xref>). This sensitivity was a significant improvement compared with previous method, which was demonstrated to be 1 in 10,000 (<xref ref-type="bibr" rid="B21">21</xref>). These results demonstrated that MIDCIRS is highly sensitive, capable of detecting a single-cell&#x02019;s amount of TCR transcripts, and rare clones could be readily and robustly detected. Those single-copy clones (minimum two identical reads) we discovered are thus likely to come from single cells (Figure <xref ref-type="fig" rid="F2">2</xref>C; Figure S6A in Supplementary Material).</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Spike-in Jurkat T cell receptor (TCR) RNA detection in na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th valign="top" align="left">Sample</th>
<th valign="top" align="center">Jurkat TCR copies detected</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">20,000Tn_10%RNA</td>
<td align="center" valign="top">7</td>
</tr>
<tr>
<td align="left" valign="top">20,000Tn_30%RNA</td>
<td align="center" valign="top">0</td>
</tr>
<tr>
<td align="left" valign="top">20,000Tn_50%RNA</td>
<td align="center" valign="top">1</td>
</tr>
<tr>
<td align="left" valign="top">100,000Tn_10%RNA</td>
<td align="center" valign="top">5</td>
</tr>
<tr>
<td align="left" valign="top">100,000Tn_30%RNA</td>
<td align="center" valign="top">4</td>
</tr>
<tr>
<td align="left" valign="top">100,000Tn_50%RNA</td>
<td align="center" valign="top">1</td>
</tr>
<tr>
<td align="left" valign="top">200,000Tn_10%RNA</td>
<td align="center" valign="top">7</td>
</tr>
<tr>
<td align="left" valign="top">200,000Tn_30%RNA</td>
<td align="center" valign="top">3</td>
</tr>
<tr>
<td align="left" valign="top">200,000Tn_50%RNA</td>
<td align="center" valign="top">3</td>
</tr>
<tr>
<td align="left" valign="top">1,000,000Tn_10%RNA</td>
<td align="center" valign="top">4</td>
</tr>
<tr>
<td align="left" valign="top">1,000,000Tn_30%RNA</td>
<td align="center" valign="top">8</td>
</tr>
<tr>
<td align="left" valign="top">1,000,000Tn_50%RNA</td>
<td align="center" valign="top">17</td>
</tr>
</tbody>
</table>
<table-wrap-foot><p><italic>10 TCR-copy worth of Jurkat RNA was added to each sample during the reverse transcription step. Number of molecular identifiers for RNA molecules that are tagged with jurkat TCR sequences were counted</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>Meanwhile, we compared the sensitivity of MIDCIRS and 5&#x02032;RACE protocol using the diversity coverage as the parameter. Briefly, the 5&#x02032;RACE protocol that was used in Smart-seq2 protocol was used for TCR-seq, which has been demonstrated to significantly improve RNA capture efficiency (<xref ref-type="bibr" rid="B38">38</xref>). Equal amount of RNA (20%) from same purification was used for both MIDCIRS and 5&#x02032;RACE protocol. We then processed sequencing results with MIDCIRS-TCR pipeline and found that 5&#x02032;RACE protocol only recovered about 44% of diversity compared to what MIDCIRS protocol obtained (Table S3 in Supplementary Material). With improved accuracy and sensitivity to detect rare clones, MIDCIRS is promising in being applied to detect MRD after treatment.</p>
</sec>
<sec id="S3-5">
<title>Quantifying T Cell Clonal Expansion in Infection Using MIDCIRS</title>
<p>It has been shown that the clonality and quantity of T cells are strongly correlated with efficacy of therapies, such as cancer chemotherapy and antiviral therapy (<xref ref-type="bibr" rid="B20">20</xref>, <xref ref-type="bibr" rid="B39">39</xref>). Accurate quantification of diversity and abundance of T cell clones is important for application of TCR-seq in clinical settings, ranging from prognosis to treatment decision-making. However, there lacks an accurate approach to evaluate the degree of T cell clonal expansion in humans. Therefore, we applied MIDCIRS TCR-seq to examine T cell clonal expansion in infection. We sorted 20,000 and 200,000 CMVpp65-specific effector CD8<sup>&#x0002B;</sup> T cells from CMV-infected patients and used 30% of RNA input to perform TCR-seq (Table S4 in Supplementary Material). CMV pp65 peptide has been shown to be the immunodominant target of CD8<sup>&#x0002B;</sup> T cell response (<xref ref-type="bibr" rid="B40">40</xref>). TCR RNA molecules were digitally counted through MIDCIRS pipeline. We defined TCR sequences with over 20 copies of RNA molecules as expanded clones according to TCR abundance distribution comparing between na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells and CMV tetramer positive effector CD8<sup>&#x0002B;</sup> T cells (Figure <xref ref-type="fig" rid="F4">4</xref>B). Over 99% unique RNA molecules were from these expanded clones in CMVpp65-specific effector CD8<sup>&#x0002B;</sup> T cells. On the other hand, although we observed uneven clonal distribution in na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells, these expanded clones only account for less than 1% unique RNA molecules (Figure <xref ref-type="fig" rid="F4">4</xref>C). Our data showed that in CMV infection, single CMV-specific TCR clone can have about 70,000 T cell progenies in 200,000 polyclonal CMV-specific effector CD8<sup>&#x0002B;</sup> T cells (Table S4 in Supplementary Material). These polyclonal CMV-specific effector CD8<sup>&#x0002B;</sup> T cells represent about 2.6% of total CD8<sup>&#x0002B;</sup> T cells. In addition, our previous study showed that tetramer positive polyclonal CMV precursor cells existed at a frequency of 1 in 100,000 CD8<sup>&#x0002B;</sup> T cells in CMV seronegative individuals (<xref ref-type="bibr" rid="B22">22</xref>). Taking together, these results suggest that single T cell clone can have about 900-fold proliferation in infection in humans. Thus, MIDCIRS can be applied to evaluate clone size and degree of clonal expansion in viral infection.</p>
</sec>
</sec>
<sec id="S4" sec-type="discussion">
<title>Discussion</title>
<p>In this study, we applied the MIDCIRS, recently developed by our group (<xref ref-type="bibr" rid="B9">9</xref>), in T cells to demonstrate (1) the necessity of MID sub-clustering to improve accuracy of repertoire diversity estimation; (2) the accuracy of counting TCR RNA molecules <italic>via</italic> MID read-distribution based barcode correction; (3) the sensitivity of detecting a single cell in as many as one million na&#x000EF;ve T cells; and (4) the ability to quantify T cell clonal expansion due to infection in CMV-seropositive patients.</p>
<p>Previous MID-based IR-seq methods, such as MIGEC, build TCR consensus sequences by grouping MIDs (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B41">41</xref>). However, the number of target molecules could vary significantly with different sample inputs, which could be challenging for choosing the appropriate MID length to ensure that each target RNA molecule is uniquely tagged by MID. Longer MIDs are likely to decrease the reverse transcription efficiency (<xref ref-type="bibr" rid="B28">28</xref>, <xref ref-type="bibr" rid="B29">29</xref>). Thus, the MIDCIRS method offers a flexible strategy for MID-barcoded IR-seq. In addition, MIGEC triages MIDs with high diversity as ambiguous. We compared TCR diversity discovered using MIDCIRS with that of MIGEC, using MID with at least two reads as the threshold for both approaches (see <xref ref-type="sec" rid="S2">Materials and Methods</xref>) and found that MIGEC led to an underestimated TCR diversity (Figure S8 in Supplementary Material, <italic>p</italic>&#x02009;&#x0003C;&#x02009;0.001, effect size <italic>r</italic>&#x02009;&#x0003D;&#x02009;0.62). We demonstrated that using MID-based sub-clustering approach, MIDCIRS could identify new diversities, prevent chimera sequences from being built, and digitally count RNA molecules (Figure <xref ref-type="fig" rid="F1">1</xref>; Figures S2 and S3 in Supplementary Material). This corrected diversity is highly consistent with cell input numbers.</p>
<p>While MIDs are useful to correct for sequencing errors and PCR errors that occur on TCR sequences, such errors are also likely to show up on MID sequences. Although these errors do not affect TCR diversity estimation, they lead to an overestimation of transcript copies, thus misestimating TCR clone size (Figure <xref ref-type="fig" rid="F2">2</xref>; Figure S4 in Supplementary Material). We corrected MID errors based on the distribution of MID read counts under MID subgroups. With MID correction, we were able to accurately count TCR RNA molecule copy number, estimate MIDCIRS detection limit as well as detect T cell clonal expansion.</p>
<p>Noteworthy, we found uneven CDR3 clone size distribution in na&#x000EF;ve CD8<sup>&#x0002B;</sup> T cells (Figure <xref ref-type="fig" rid="F4">4</xref>B). The most expanded clone was enriched about 0.27% (Table S1 in Supplementary Material). This could be due to convergent recombination as has been previously noted (<xref ref-type="bibr" rid="B42">42</xref>, <xref ref-type="bibr" rid="B43">43</xref>) or uneven clonal expansion during thymocyte maturation and selection in thymus (<xref ref-type="bibr" rid="B44">44</xref>, <xref ref-type="bibr" rid="B45">45</xref>).</p>
<p>Furthermore, there is a lack of standard guidelines of how much RNA input to use for library preparation and sequencing. Also, the capacity to evaluate immune repertoire and gene expression profile simultaneously will facilitate clinical practice, such as cancer immunotherapies. Efforts have been made to reconstruct antibody and TCR repertoire from RNA-seq data. This, however, requires very deep sequencing to recover highly expanded T cell clones in the sample, and the exact degree of repertoire coverage is difficult to assess (<xref ref-type="bibr" rid="B46">46</xref>&#x02013;<xref ref-type="bibr" rid="B48">48</xref>). Here, we demonstrated that 50% RNA is enough to cover about 60% of CDR3 diversity (Figure <xref ref-type="fig" rid="F3">3</xref>B), making it beneficial to take advantage of the rest of the RNA from the same sample for other applications, e.g., RNA-seq.</p>
<p>Based on the TCR diversity estimation and its dependency on RNA input, we built a probability model to estimate TCR RNA molecule copies, which resulted in three copies per cell (Figure <xref ref-type="fig" rid="F3">3</xref>B). We would like to point out that this does not mean that on average there are three copies of TCR RNA in a T cell. Because of the efficiency of RNA purification and reverse transcription, we expect our observed RNA molecule per cell to be lower than the true value. In Fact, dPCR results showed an average of 10 copies of TCR RNA molecule per cell (Figure <xref ref-type="fig" rid="F3">3</xref>C), suggesting the efficiency of MIDCIRS in TCR RNA molecule digital counting is about 30%, which is consistent with previous finding that nanoliter reaction volume significantly improved PCR efficiency. Thus, quantifying TCR RNA molecule per cell enables us to estimate the extent of T cell clonal expansion that was not possible until now.</p>
<p>We also used spike-in TCR RNA to validate the sensitivity of MIDCIRS. We showed that spiked-in TCR RNA at as few as five copies can be reliably detected across multiple libraries (Figure <xref ref-type="fig" rid="F4">4</xref>A). More importantly, we were also able to detect a single-cell worth of RNA in as many as one million cells (Table <xref ref-type="table" rid="T1">1</xref>). With this demonstrated sensitivity, this method could be extremely useful in MRD detection.</p>
<p>Last, we applied MIDCIRS to evaluate T cell clonal expansion in CMV-infected patients. Through accurate digital counting of TCR RNA molecules and in combination of precursor T cell frequency, we showed that CMV-specific effector CD8<sup>&#x0002B;</sup> T cells can expand at least 900 times, and there could be more than 70,000 effector CD8<sup>&#x0002B;</sup> T cells derived from the same CMV-specific T cell clone in total of 7,700,000 of CD8<sup>&#x0002B;</sup> T cell in infection. We also noticed that there is a potential of same TCR sequences tagged with same MID, which would under estimate the clonal size, especially in highly expanded clones. We calculated the expected number of collisions where same MIDs tag same RNA molecules (Supplementary Methods in Supplementary Material). With MID length being 12, when there are 200,000 identical RNA molecules, the percentage of identical RNA molecules tagged with same MID is only 1%. While long MID decreases the percentage of identical RNA molecules tagged with same MID, it also decreases efficiency of reverse transcription. Our analysis revealed that MID with 12 nucleotides is appropriate. Therefore, MIDCIRS provides the foundation of accurate assessment of clone size and clonal expansion in infection and vaccination, which would be a useful technology to provide a comprehensive quantification of the T cell repertoire in various basic studies and clinical settings.</p>
</sec>
<sec id="S5">
<title>Ethics Statement</title>
<p>The protocol of using de-identified blood donors&#x02019; sample was approved by the IRB board of University of Texas at Austin.</p>
</sec>
<sec id="S6">
<title>Data Access</title>
<p>All sequencing data are under SRA accession SRP128082.</p>
</sec>
<sec id="S7" sec-type="author-contributor">
<title>Author Contributions</title>
<p>K-YM performed all library preparation, data analysis, and wrote the manuscript; CH developed MIDCIRS-TCR analysis pipeline and RNA copy number simulation model; BW helped with na&#x000EF;ve T cell sorting and manuscript editing; CW helped with CMV-specific T cell sorting and CMV-specific T cell line culture; JX helped to optimize MIDCIRS pipeline. HY helped with sequencing. NJ conceived the idea, designed the study, directed data analysis, and revised the manuscript with contributions from all coauthors.</p>
</sec>
<sec id="S8">
<title>Disclaimer</title>
<p>The protocol of using de-identified blood donors&#x02019; sample was approved by the IRB board of University of Texas at Austin.</p>
</sec>
<sec id="S9">
<title>Conflict of Interest Statement</title>
<p>NJ is a scientific advisor of ImmuDX, LLC. A provisional patent application has been filed by the University of Texas at Austin on the method described here.</p>
</sec>
</body>
<back>
<ack>
<p>The authors would like to thank We Are Blood (Austin, TX, USA), for providing the blood samples, Jessica Podnar, and Dr. Michael Wilson at the Genomic Sequencing and Analysis Facility at UT Austin for helping with the sequencing runs.</p>
</ack>
<fn-group>
<fn fn-type="financial-disclosure">
<p><bold>Funding.</bold> This work was supported by NIH grants R00AG040149 (NJ) and S10OD020072 (NJ), NSF CAREER Award 1653866 (NJ), the Welch Foundation grant F1785 (NJ), and National Natural Science Foundation of China grants 1147222 and 11672246 (HY). NJ is a Cancer Prevention and Research Institute of Texas (CPRIT) Scholar and a Damon Runyon-Rachleff Innovator. BW is a recipient of the Thrust 2000&#x02014;George Sawyer Endowed Graduate Fellowship in Engineering.</p></fn>
</fn-group>
<sec id="S10" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at <uri xlink:href="http://www.frontiersin.org/articles/10.3389/fimmu.2018.00033/full&#x00023;supplementary-material">http://www.frontiersin.org/articles/10.3389/fimmu.2018.00033/full&#x00023;supplementary-material</uri>.</p>
<supplementary-material xlink:href="data_sheet_1.PDF" id="SM1" mimetype="applicationn/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><label>1</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ellebedy</surname> <given-names>AH</given-names></name> <name><surname>Jackson</surname> <given-names>KJ</given-names></name> <name><surname>Kissick</surname> <given-names>HT</given-names></name> <name><surname>Nakaya</surname> <given-names>HI</given-names></name> <name><surname>Davis</surname> <given-names>CW</given-names></name> <name><surname>Roskin</surname> <given-names>KM</given-names></name> <etal/></person-group> <article-title>Defining antigen-specific plasmablast and memory B cell subsets in human blood after viral infection or vaccination</article-title>. <source>Nat Immunol</source> (<year>2016</year>) <volume>17</volume>(<issue>10</issue>):<fpage>1226</fpage>&#x02013;<lpage>34</lpage>.<pub-id pub-id-type="doi">10.1038/ni.3533</pub-id><pub-id pub-id-type="pmid">27525369</pub-id></citation></ref>
<ref id="B2"><label>2</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>N</given-names></name> <name><surname>He</surname> <given-names>J</given-names></name> <name><surname>Weinstein</surname> <given-names>JA</given-names></name> <name><surname>Penland</surname> <given-names>L</given-names></name> <name><surname>Sasaki</surname> <given-names>S</given-names></name> <name><surname>He</surname> <given-names>XS</given-names></name> <etal/></person-group> <article-title>Lineage structure of the human antibody repertoire in response to influenza vaccination</article-title>. <source>Sci Transl Med</source> (<year>2013</year>) <volume>5</volume>(<issue>171</issue>):<fpage>171ra19</fpage>.<pub-id pub-id-type="doi">10.1126/scitranslmed.3004794</pub-id><pub-id pub-id-type="pmid">23390249</pub-id></citation></ref>
<ref id="B3"><label>3</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>J</given-names></name> <name><surname>Boutz</surname> <given-names>DR</given-names></name> <name><surname>Chromikova</surname> <given-names>V</given-names></name> <name><surname>Joyce</surname> <given-names>MG</given-names></name> <name><surname>Vollmers</surname> <given-names>C</given-names></name> <name><surname>Leung</surname> <given-names>K</given-names></name> <etal/></person-group> <article-title>Molecular-level analysis of the serum antibody repertoire in young adults before and after seasonal influenza vaccination</article-title>. <source>Nat Med</source> (<year>2016</year>) <volume>22</volume>(<issue>12</issue>):<fpage>1456</fpage>&#x02013;<lpage>64</lpage>.<pub-id pub-id-type="doi">10.1038/nm.4224</pub-id><pub-id pub-id-type="pmid">27820605</pub-id></citation></ref>
<ref id="B4"><label>4</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>IJspeert</surname> <given-names>H</given-names></name> <name><surname>van Schouwenburg</surname> <given-names>PA</given-names></name> <name><surname>van Zessen</surname> <given-names>D</given-names></name> <name><surname>Pico-Knijnenburg</surname> <given-names>I</given-names></name> <name><surname>Driessen</surname> <given-names>GJ</given-names></name> <name><surname>Stubbs</surname> <given-names>AP</given-names></name> <etal/></person-group> <article-title>Evaluation of the antigen-experienced B-cell receptor repertoire in healthy children and adults</article-title>. <source>Front Immunol</source> (<year>2016</year>) <volume>7</volume>:<fpage>410</fpage>.<pub-id pub-id-type="doi">10.3389/fimmu.2016.00410</pub-id><pub-id pub-id-type="pmid">27799928</pub-id></citation></ref>
<ref id="B5"><label>5</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>N</given-names></name> <name><surname>Weinstein</surname> <given-names>JA</given-names></name> <name><surname>Penland</surname> <given-names>L</given-names></name> <name><surname>White</surname> <given-names>RA</given-names> <suffix>III</suffix></name> <name><surname>Fisher</surname> <given-names>DS</given-names></name> <name><surname>Quake</surname> <given-names>SR</given-names></name></person-group>. <article-title>Determinism and stochasticity during maturation of the zebrafish antibody repertoire</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2011</year>) <volume>108</volume>(<issue>13</issue>):<fpage>5348</fpage>&#x02013;<lpage>53</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.1014277108</pub-id><pub-id pub-id-type="pmid">21393572</pub-id></citation></ref>
<ref id="B6"><label>6</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Prabakaran</surname> <given-names>P</given-names></name> <name><surname>Chen</surname> <given-names>W</given-names></name> <name><surname>Singarayan</surname> <given-names>MG</given-names></name> <name><surname>Stewart</surname> <given-names>CC</given-names></name> <name><surname>Streaker</surname> <given-names>E</given-names></name> <name><surname>Feng</surname> <given-names>Y</given-names></name> <etal/></person-group> <article-title>Expressed antibody repertoires in human cord blood cells: 454 sequencing and IMGT/HighV-QUEST analysis of germline gene usage, junctional diversity, and somatic mutations</article-title>. <source>Immunogenetics</source> (<year>2012</year>) <volume>64</volume>(<issue>5</issue>):<fpage>337</fpage>&#x02013;<lpage>50</lpage>.<pub-id pub-id-type="doi">10.1007/s00251-011-0595-8</pub-id><pub-id pub-id-type="pmid">22200891</pub-id></citation></ref>
<ref id="B7"><label>7</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rechavi</surname> <given-names>E</given-names></name> <name><surname>Lev</surname> <given-names>A</given-names></name> <name><surname>Lee</surname> <given-names>YN</given-names></name> <name><surname>Simon</surname> <given-names>AJ</given-names></name> <name><surname>Yinon</surname> <given-names>Y</given-names></name> <name><surname>Lipitz</surname> <given-names>S</given-names></name> <etal/></person-group> <article-title>Timely and spatially regulated maturation of B and T cell repertoire during human fetal development</article-title>. <source>Sci Transl Med</source> (<year>2015</year>) <volume>7</volume>(<issue>276</issue>):<fpage>276ra25</fpage>.<pub-id pub-id-type="doi">10.1126/scitranslmed.aaa0072</pub-id><pub-id pub-id-type="pmid">25717098</pub-id></citation></ref>
<ref id="B8"><label>8</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weinstein</surname> <given-names>JA</given-names></name> <name><surname>Jiang</surname> <given-names>N</given-names></name> <name><surname>White</surname> <given-names>RA</given-names> <suffix>III</suffix></name> <name><surname>Fisher</surname> <given-names>DS</given-names></name> <name><surname>Quake</surname> <given-names>SR</given-names></name></person-group>. <article-title>High-throughput sequencing of the zebrafish antibody repertoire</article-title>. <source>Science</source> (<year>2009</year>) <volume>324</volume>(<issue>5928</issue>):<fpage>807</fpage>&#x02013;<lpage>10</lpage>.<pub-id pub-id-type="doi">10.1126/science.1170020</pub-id><pub-id pub-id-type="pmid">19423829</pub-id></citation></ref>
<ref id="B9"><label>9</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wendel</surname> <given-names>BS</given-names></name> <name><surname>He</surname> <given-names>C</given-names></name> <name><surname>Qu</surname> <given-names>M</given-names></name> <name><surname>Wu</surname> <given-names>D</given-names></name> <name><surname>Hernandez</surname> <given-names>SM</given-names></name> <name><surname>Ma</surname> <given-names>KY</given-names></name> <etal/></person-group> <article-title>Accurate immune repertoire sequencing reveals malaria infection driven antibody lineage diversification in young children</article-title>. <source>Nat Commun</source> (<year>2017</year>) <volume>8</volume>(<issue>1</issue>):<fpage>531</fpage>.<pub-id pub-id-type="doi">10.1038/s41467-017-00645-x</pub-id><pub-id pub-id-type="pmid">28912592</pub-id></citation></ref>
<ref id="B10"><label>10</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Faham</surname> <given-names>M</given-names></name> <name><surname>Zheng</surname> <given-names>J</given-names></name> <name><surname>Moorhead</surname> <given-names>M</given-names></name> <name><surname>Carlton</surname> <given-names>VE</given-names></name> <name><surname>Stow</surname> <given-names>P</given-names></name> <name><surname>Coustan-Smith</surname> <given-names>E</given-names></name> <etal/></person-group> <article-title>Deep-sequencing approach for minimal residual disease detection in acute lymphoblastic leukemia</article-title>. <source>Blood</source> (<year>2012</year>) <volume>120</volume>(<issue>26</issue>):<fpage>5173</fpage>&#x02013;<lpage>80</lpage>.<pub-id pub-id-type="doi">10.1182/blood-2012-07-444042</pub-id><pub-id pub-id-type="pmid">23074282</pub-id></citation></ref>
<ref id="B11"><label>11</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Logan</surname> <given-names>AC</given-names></name> <name><surname>Gao</surname> <given-names>H</given-names></name> <name><surname>Wang</surname> <given-names>C</given-names></name> <name><surname>Sahaf</surname> <given-names>B</given-names></name> <name><surname>Jones</surname> <given-names>CD</given-names></name> <name><surname>Marshall</surname> <given-names>EL</given-names></name> <etal/></person-group> <article-title>High-throughput VDJ sequencing for quantification of minimal residual disease in chronic lymphocytic leukemia and immune reconstitution assessment</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2011</year>) <volume>108</volume>(<issue>52</issue>):<fpage>21194</fpage>&#x02013;<lpage>9</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.1118357109</pub-id><pub-id pub-id-type="pmid">22160699</pub-id></citation></ref>
<ref id="B12"><label>12</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>AC</given-names></name> <name><surname>Postow</surname> <given-names>MA</given-names></name> <name><surname>Orlowski</surname> <given-names>RJ</given-names></name> <name><surname>Mick</surname> <given-names>R</given-names></name> <name><surname>Bengsch</surname> <given-names>B</given-names></name> <name><surname>Manne</surname> <given-names>S</given-names></name> <etal/></person-group> <article-title>T-cell invigoration to tumour burden ratio associated with anti-PD-1 response</article-title>. <source>Nature</source> (<year>2017</year>) <volume>545</volume>(<issue>7652</issue>):<fpage>60</fpage>&#x02013;<lpage>5</lpage>.<pub-id pub-id-type="doi">10.1038/nature22079</pub-id><pub-id pub-id-type="pmid">28397821</pub-id></citation></ref>
<ref id="B13"><label>13</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>Q</given-names></name> <name><surname>Zhou</surname> <given-names>J</given-names></name> <name><surname>Chen</surname> <given-names>G</given-names></name> <name><surname>Shi</surname> <given-names>Y</given-names></name> <name><surname>Yu</surname> <given-names>H</given-names></name> <name><surname>Guan</surname> <given-names>P</given-names></name> <etal/></person-group> <article-title>Diversity index of mucosal resident T lymphocyte repertoire predicts clinical prognosis in gastric cancer</article-title>. <source>Oncoimmunology</source> (<year>2015</year>) <volume>4</volume>(<issue>4</issue>):<fpage>e1001230</fpage>.<pub-id pub-id-type="doi">10.1080/2162402X.2014.1001230</pub-id><pub-id pub-id-type="pmid">26137399</pub-id></citation></ref>
<ref id="B14"><label>14</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khan</surname> <given-names>TA</given-names></name> <name><surname>Friedensohn</surname> <given-names>S</given-names></name> <name><surname>Gorter de Vries</surname> <given-names>AR</given-names></name> <name><surname>Straszewski</surname> <given-names>J</given-names></name> <name><surname>Ruscheweyh</surname> <given-names>HJ</given-names></name> <name><surname>Reddy</surname> <given-names>ST</given-names></name></person-group>. <article-title>Accurate and predictive antibody repertoire profiling by molecular amplification fingerprinting</article-title>. <source>Sci Adv</source> (<year>2016</year>) <volume>2</volume>(<issue>3</issue>):<fpage>e1501371</fpage>.<pub-id pub-id-type="doi">10.1126/sciadv.1501371</pub-id><pub-id pub-id-type="pmid">26998518</pub-id></citation></ref>
<ref id="B15"><label>15</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kinde</surname> <given-names>I</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Papadopoulos</surname> <given-names>N</given-names></name> <name><surname>Kinzler</surname> <given-names>KW</given-names></name> <name><surname>Vogelstein</surname> <given-names>B</given-names></name></person-group>. <article-title>Detection and quantification of rare mutations with massively parallel sequencing</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2011</year>) <volume>108</volume>(<issue>23</issue>):<fpage>9530</fpage>&#x02013;<lpage>5</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.1105422108</pub-id><pub-id pub-id-type="pmid">21586637</pub-id></citation></ref>
<ref id="B16"><label>16</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shugay</surname> <given-names>M</given-names></name> <name><surname>Bagaev</surname> <given-names>DV</given-names></name> <name><surname>Turchaninova</surname> <given-names>MA</given-names></name> <name><surname>Bolotin</surname> <given-names>DA</given-names></name> <name><surname>Britanova</surname> <given-names>OV</given-names></name> <name><surname>Putintseva</surname> <given-names>EV</given-names></name> <etal/></person-group> <article-title>VDJtools: unifying post-analysis of T cell receptor repertoires</article-title>. <source>PLoS Comput Biol</source> (<year>2015</year>) <volume>11</volume>(<issue>11</issue>):<fpage>e1004503</fpage>.<pub-id pub-id-type="doi">10.1371/journal.pcbi.1004503</pub-id><pub-id pub-id-type="pmid">26606115</pub-id></citation></ref>
<ref id="B17"><label>17</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shugay</surname> <given-names>M</given-names></name> <name><surname>Britanova</surname> <given-names>OV</given-names></name> <name><surname>Merzlyak</surname> <given-names>EM</given-names></name> <name><surname>Turchaninova</surname> <given-names>MA</given-names></name> <name><surname>Mamedov</surname> <given-names>IZ</given-names></name> <name><surname>Tuganbaev</surname> <given-names>TR</given-names></name> <etal/></person-group> <article-title>Towards error-free profiling of immune repertoires</article-title>. <source>Nat Methods</source> (<year>2014</year>) <volume>11</volume>(<issue>6</issue>):<fpage>653</fpage>&#x02013;<lpage>5</lpage>.<pub-id pub-id-type="doi">10.1038/nmeth.2960</pub-id><pub-id pub-id-type="pmid">24793455</pub-id></citation></ref>
<ref id="B18"><label>18</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vander Heiden</surname> <given-names>JA</given-names></name> <name><surname>Yaari</surname> <given-names>G</given-names></name> <name><surname>Uduman</surname> <given-names>M</given-names></name> <name><surname>Stern</surname> <given-names>JN</given-names></name> <name><surname>O&#x02019;Connor</surname> <given-names>KC</given-names></name> <name><surname>Hafler</surname> <given-names>DA</given-names></name> <etal/></person-group> <article-title>pRESTO: a toolkit for processing high-throughput sequencing raw reads of lymphocyte receptor repertoires</article-title>. <source>Bioinformatics</source> (<year>2014</year>) <volume>30</volume>(<issue>13</issue>):<fpage>1930</fpage>&#x02013;<lpage>2</lpage>.<pub-id pub-id-type="doi">10.1093/bioinformatics/btu138</pub-id><pub-id pub-id-type="pmid">24618469</pub-id></citation></ref>
<ref id="B19"><label>19</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vollmers</surname> <given-names>C</given-names></name> <name><surname>Sit</surname> <given-names>RV</given-names></name> <name><surname>Weinstein</surname> <given-names>JA</given-names></name> <name><surname>Dekker</surname> <given-names>CL</given-names></name> <name><surname>Quake</surname> <given-names>SR</given-names></name></person-group>. <article-title>Genetic measurement of memory B-cell recall using antibody repertoire sequencing</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2013</year>) <volume>110</volume>(<issue>33</issue>):<fpage>13463</fpage>&#x02013;<lpage>8</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.1312146110</pub-id><pub-id pub-id-type="pmid">23898164</pub-id></citation></ref>
<ref id="B20"><label>20</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Robins</surname> <given-names>HS</given-names></name> <name><surname>Ericson</surname> <given-names>NG</given-names></name> <name><surname>Guenthoer</surname> <given-names>J</given-names></name> <name><surname>O&#x02019;Briant</surname> <given-names>KC</given-names></name> <name><surname>Tewari</surname> <given-names>M</given-names></name> <name><surname>Drescher</surname> <given-names>CW</given-names></name> <etal/></person-group> <article-title>Digital genomic quantification of tumor-infiltrating lymphocytes</article-title>. <source>Sci Transl Med</source> (<year>2013</year>) <volume>5</volume>(<issue>214</issue>):<fpage>214ra169</fpage>.<pub-id pub-id-type="doi">10.1126/scitranslmed.3007247</pub-id><pub-id pub-id-type="pmid">24307693</pub-id></citation></ref>
<ref id="B21"><label>21</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ruggiero</surname> <given-names>E</given-names></name> <name><surname>Nicolay</surname> <given-names>JP</given-names></name> <name><surname>Fronza</surname> <given-names>R</given-names></name> <name><surname>Arens</surname> <given-names>A</given-names></name> <name><surname>Paruzynski</surname> <given-names>A</given-names></name> <name><surname>Nowrouzi</surname> <given-names>A</given-names></name> <etal/></person-group> <article-title>High-resolution analysis of the human T-cell receptor repertoire</article-title>. <source>Nat Commun</source> (<year>2015</year>) <volume>6</volume>:<fpage>8081</fpage>.<pub-id pub-id-type="doi">10.1038/ncomms9081</pub-id><pub-id pub-id-type="pmid">26324409</pub-id></citation></ref>
<ref id="B22"><label>22</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>W</given-names></name> <name><surname>Jiang</surname> <given-names>N</given-names></name> <name><surname>Ebert</surname> <given-names>PJ</given-names></name> <name><surname>Kidd</surname> <given-names>BA</given-names></name> <name><surname>Muller</surname> <given-names>S</given-names></name> <name><surname>Lund</surname> <given-names>PJ</given-names></name> <etal/></person-group> <article-title>Clonal deletion prunes but does not eliminate self-specific alphabeta CD8(&#x0002B;) T lymphocytes</article-title>. <source>Immunity</source> (<year>2015</year>) <volume>42</volume>(<issue>5</issue>):<fpage>929</fpage>&#x02013;<lpage>41</lpage>.<pub-id pub-id-type="doi">10.1016/j.immuni.2015.05.001</pub-id></citation></ref>
<ref id="B23"><label>23</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>SQ</given-names></name> <name><surname>Parker</surname> <given-names>P</given-names></name> <name><surname>Ma</surname> <given-names>KY</given-names></name> <name><surname>He</surname> <given-names>C</given-names></name> <name><surname>Shi</surname> <given-names>Q</given-names></name> <name><surname>Cui</surname> <given-names>Z</given-names></name> <etal/></person-group> <article-title>Direct measurement of T cell receptor affinity and sequence from naive antiviral T cells</article-title>. <source>Sci Transl Med</source> (<year>2016</year>) <volume>8</volume>(<issue>341</issue>):<fpage>341ra77</fpage>.<pub-id pub-id-type="doi">10.1126/scitranslmed.aaf1278</pub-id></citation></ref>
<ref id="B24"><label>24</label><citation citation-type="web"><person-group person-group-type="author"><name><surname>Mora</surname> <given-names>T</given-names></name> <name><surname>Walczak</surname> <given-names>A</given-names></name></person-group>. <article-title>Quantifying lymphocyte receptor diversity</article-title>. <source>ArXiv e-prints</source> (<year>2016</year>) 1604. Available from: <uri xlink:href="http://adsabs.harvard.edu/abs/2016arXiv160400487M">http://adsabs.harvard.edu/abs/2016arXiv160400487M</uri></citation></ref>
<ref id="B25"><label>25</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clauset</surname> <given-names>A</given-names></name> <name><surname>Shalizi</surname> <given-names>CR</given-names></name> <name><surname>Newman</surname> <given-names>MEJ</given-names></name></person-group>. <article-title>Power-law distributions in empirical data</article-title>. <source>SIAM Rev</source> (<year>2009</year>) <volume>51</volume>(<issue>4</issue>):<fpage>661</fpage>&#x02013;<lpage>703</lpage>.<pub-id pub-id-type="doi">10.1137/070710111</pub-id></citation></ref>
<ref id="B26"><label>26</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Csardi</surname> <given-names>G</given-names></name> <name><surname>Nepusz</surname> <given-names>T</given-names></name></person-group>. <article-title>The igraph software package for complex network research</article-title>. <source>Int J Complex Syst</source> (<year>2006</year>) <volume>1695</volume>(<issue>5</issue>):<fpage>1</fpage>&#x02013;<lpage>9</lpage>.</citation></ref>
<ref id="B27"><label>27</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Briney</surname> <given-names>B</given-names></name> <name><surname>Le</surname> <given-names>K</given-names></name> <name><surname>Zhu</surname> <given-names>J</given-names></name> <name><surname>Burton</surname> <given-names>DR</given-names></name></person-group>. <article-title>Clonify: unseeded antibody lineage assignment from next-generation sequencing data</article-title>. <source>Sci Rep</source> (<year>2016</year>) <volume>6</volume>:<fpage>23901</fpage>.<pub-id pub-id-type="doi">10.1038/srep23901</pub-id><pub-id pub-id-type="pmid">27102563</pub-id></citation></ref>
<ref id="B28"><label>28</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shiao</surname> <given-names>YH</given-names></name></person-group>. <article-title>A new reverse transcription-polymerase chain reaction method for accurate quantification</article-title>. <source>BMC Biotechnol</source> (<year>2003</year>) <volume>3</volume>:<fpage>22</fpage>.<pub-id pub-id-type="doi">10.1186/1472-6750-3-22</pub-id><pub-id pub-id-type="pmid">14664723</pub-id></citation></ref>
<ref id="B29"><label>29</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zajac</surname> <given-names>P</given-names></name> <name><surname>Islam</surname> <given-names>S</given-names></name> <name><surname>Hochgerner</surname> <given-names>H</given-names></name> <name><surname>Lonnerberg</surname> <given-names>P</given-names></name> <name><surname>Linnarsson</surname> <given-names>S</given-names></name></person-group>. <article-title>Base preferences in non-templated nucleotide incorporation by MMLV-derived reverse transcriptases</article-title>. <source>PLoS One</source> (<year>2013</year>) <volume>8</volume>(<issue>12</issue>):<fpage>e85270</fpage>.<pub-id pub-id-type="doi">10.1371/journal.pone.0085270</pub-id><pub-id pub-id-type="pmid">24392002</pub-id></citation></ref>
<ref id="B30"><label>30</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>HC</given-names></name> <name><surname>Fu</surname> <given-names>GK</given-names></name> <name><surname>Fodor</surname> <given-names>SP</given-names></name></person-group>. <article-title>Combinatorial labeling of single cells for gene expression cytometry</article-title>. <source>Science</source> (<year>2015</year>) <volume>347</volume>(<issue>6222</issue>):<fpage>1258367</fpage>.<pub-id pub-id-type="doi">10.1126/science.1258367</pub-id></citation></ref>
<ref id="B31"><label>31</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>GK</given-names></name> <name><surname>Hu</surname> <given-names>J</given-names></name> <name><surname>Wang</surname> <given-names>PH</given-names></name> <name><surname>Fodor</surname> <given-names>SP</given-names></name></person-group>. <article-title>Counting individual DNA molecules by the stochastic attachment of diverse labels</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2011</year>) <volume>108</volume>(<issue>22</issue>):<fpage>9026</fpage>&#x02013;<lpage>31</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.1017621108</pub-id><pub-id pub-id-type="pmid">21562209</pub-id></citation></ref>
<ref id="B32"><label>32</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Islam</surname> <given-names>S</given-names></name> <name><surname>Zeisel</surname> <given-names>A</given-names></name> <name><surname>Joost</surname> <given-names>S</given-names></name> <name><surname>La Manno</surname> <given-names>G</given-names></name> <name><surname>Zajac</surname> <given-names>P</given-names></name> <name><surname>Kasper</surname> <given-names>M</given-names></name> <etal/></person-group> <article-title>Quantitative single-cell RNA-seq with unique molecular identifiers</article-title>. <source>Nat Methods</source> (<year>2014</year>) <volume>11</volume>(<issue>2</issue>):<fpage>163</fpage>&#x02013;<lpage>6</lpage>.<pub-id pub-id-type="doi">10.1038/nmeth.2772</pub-id><pub-id pub-id-type="pmid">24363023</pub-id></citation></ref>
<ref id="B33"><label>33</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shiroguchi</surname> <given-names>K</given-names></name> <name><surname>Jia</surname> <given-names>TZ</given-names></name> <name><surname>Sims</surname> <given-names>PA</given-names></name> <name><surname>Xie</surname> <given-names>XS</given-names></name></person-group>. <article-title>Digital RNA sequencing minimizes sequence-dependent bias and amplification noise with optimized single-molecule barcodes</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2012</year>) <volume>109</volume>(<issue>4</issue>):<fpage>1347</fpage>&#x02013;<lpage>52</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.1118018109</pub-id><pub-id pub-id-type="pmid">22232676</pub-id></citation></ref>
<ref id="B34"><label>34</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>GK</given-names></name> <name><surname>Wilhelmy</surname> <given-names>J</given-names></name> <name><surname>Stern</surname> <given-names>D</given-names></name> <name><surname>Fan</surname> <given-names>HC</given-names></name> <name><surname>Fodor</surname> <given-names>SP</given-names></name></person-group>. <article-title>Digital encoding of cellular mRNAs enabling precise and absolute gene expression measurement by single-molecule counting</article-title>. <source>Anal Chem</source> (<year>2014</year>) <volume>86</volume>(<issue>6</issue>):<fpage>2867</fpage>&#x02013;<lpage>70</lpage>.<pub-id pub-id-type="doi">10.1021/ac500459p</pub-id><pub-id pub-id-type="pmid">24579851</pub-id></citation></ref>
<ref id="B35"><label>35</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>W</given-names></name> <name><surname>Liao</surname> <given-names>Y</given-names></name> <name><surname>Willis</surname> <given-names>SN</given-names></name> <name><surname>Taubenheim</surname> <given-names>N</given-names></name> <name><surname>Inouye</surname> <given-names>M</given-names></name> <name><surname>Tarlinton</surname> <given-names>DM</given-names></name> <etal/></person-group> <article-title>Transcriptional profiling of mouse B cell terminal differentiation defines a signature for antibody-secreting plasma cells</article-title>. <source>Nat Immunol</source> (<year>2015</year>) <volume>16</volume>(<issue>6</issue>):<fpage>663</fpage>&#x02013;<lpage>73</lpage>.<pub-id pub-id-type="doi">10.1038/ni.3154</pub-id><pub-id pub-id-type="pmid">25894659</pub-id></citation></ref>
<ref id="B36"><label>36</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Warren</surname> <given-names>L</given-names></name> <name><surname>Bryder</surname> <given-names>D</given-names></name> <name><surname>Weissman</surname> <given-names>IL</given-names></name> <name><surname>Quake</surname> <given-names>SR</given-names></name></person-group>. <article-title>Transcription factor profiling in individual hematopoietic progenitors by digital RT-PCR</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2006</year>) <volume>103</volume>(<issue>47</issue>):<fpage>17807</fpage>&#x02013;<lpage>12</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.0608512103</pub-id><pub-id pub-id-type="pmid">17098862</pub-id></citation></ref>
<ref id="B37"><label>37</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Warren</surname> <given-names>RL</given-names></name> <name><surname>Freeman</surname> <given-names>JD</given-names></name> <name><surname>Zeng</surname> <given-names>T</given-names></name> <name><surname>Choe</surname> <given-names>G</given-names></name> <name><surname>Munro</surname> <given-names>S</given-names></name> <name><surname>Moore</surname> <given-names>R</given-names></name> <etal/></person-group> <article-title>Exhaustive T-cell repertoire sequencing of human peripheral blood samples reveals signatures of antigen selection and a directly measured repertoire size of at least 1 million clonotypes</article-title>. <source>Genome Res</source> (<year>2011</year>) <volume>21</volume>(<issue>5</issue>):<fpage>790</fpage>&#x02013;<lpage>7</lpage>.<pub-id pub-id-type="doi">10.1101/gr.115428.110</pub-id><pub-id pub-id-type="pmid">21349924</pub-id></citation></ref>
<ref id="B38"><label>38</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Picelli</surname> <given-names>S</given-names></name> <name><surname>Bjorklund</surname> <given-names>AK</given-names></name> <name><surname>Faridani</surname> <given-names>OR</given-names></name> <name><surname>Sagasser</surname> <given-names>S</given-names></name> <name><surname>Winberg</surname> <given-names>G</given-names></name> <name><surname>Sandberg</surname> <given-names>R</given-names></name></person-group>. <article-title>Smart-seq2 for sensitive full-length transcriptome profiling in single cells</article-title>. <source>Nat Methods</source> (<year>2013</year>) <volume>10</volume>(<issue>11</issue>):<fpage>1096</fpage>&#x02013;<lpage>8</lpage>.<pub-id pub-id-type="doi">10.1038/nmeth.2639</pub-id><pub-id pub-id-type="pmid">24056875</pub-id></citation></ref>
<ref id="B39"><label>39</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Heather</surname> <given-names>JM</given-names></name> <name><surname>Best</surname> <given-names>K</given-names></name> <name><surname>Oakes</surname> <given-names>T</given-names></name> <name><surname>Gray</surname> <given-names>ER</given-names></name> <name><surname>Roe</surname> <given-names>JK</given-names></name> <name><surname>Thomas</surname> <given-names>N</given-names></name> <etal/></person-group> <article-title>Dynamic perturbations of the T-cell receptor repertoire in chronic HIV infection and following antiretroviral therapy</article-title>. <source>Front Immunol</source> (<year>2015</year>) <volume>6</volume>:<fpage>644</fpage>.<pub-id pub-id-type="doi">10.3389/fimmu.2015.00644</pub-id></citation></ref>
<ref id="B40"><label>40</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wills</surname> <given-names>MR</given-names></name> <name><surname>Carmichael</surname> <given-names>AJ</given-names></name> <name><surname>Mynard</surname> <given-names>K</given-names></name> <name><surname>Jin</surname> <given-names>X</given-names></name> <name><surname>Weekes</surname> <given-names>MP</given-names></name> <name><surname>Plachter</surname> <given-names>B</given-names></name> <etal/></person-group> <article-title>The human cytotoxic T-lymphocyte (CTL) response to cytomegalovirus is dominated by structural protein pp65: frequency, specificity, and T-cell receptor usage of pp65-specific CTL</article-title>. <source>J Virol</source> (<year>1996</year>) <volume>70</volume>(<issue>11</issue>):<fpage>7569</fpage>&#x02013;<lpage>79</lpage>.<pub-id pub-id-type="pmid">8892876</pub-id></citation></ref>
<ref id="B41"><label>41</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Egorov</surname> <given-names>ES</given-names></name> <name><surname>Merzlyak</surname> <given-names>EM</given-names></name> <name><surname>Shelenkov</surname> <given-names>AA</given-names></name> <name><surname>Britanova</surname> <given-names>OV</given-names></name> <name><surname>Sharonov</surname> <given-names>GV</given-names></name> <name><surname>Staroverov</surname> <given-names>DB</given-names></name> <etal/></person-group> <article-title>Quantitative profiling of immune repertoires for minor lymphocyte counts using unique molecular identifiers</article-title>. <source>J Immunol</source> (<year>2015</year>) <volume>194</volume>(<issue>12</issue>):<fpage>6155</fpage>&#x02013;<lpage>63</lpage>.<pub-id pub-id-type="doi">10.4049/jimmunol.1500215</pub-id><pub-id pub-id-type="pmid">25957172</pub-id></citation></ref>
<ref id="B42"><label>42</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Quigley</surname> <given-names>MF</given-names></name> <name><surname>Greenaway</surname> <given-names>HY</given-names></name> <name><surname>Venturi</surname> <given-names>V</given-names></name> <name><surname>Lindsay</surname> <given-names>R</given-names></name> <name><surname>Quinn</surname> <given-names>KM</given-names></name> <name><surname>Seder</surname> <given-names>RA</given-names></name> <etal/></person-group> <article-title>Convergent recombination shapes the clonotypic landscape of the naive T-cell repertoire</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2010</year>) <volume>107</volume>(<issue>45</issue>):<fpage>19414</fpage>&#x02013;<lpage>9</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.1010586107</pub-id><pub-id pub-id-type="pmid">20974936</pub-id></citation></ref>
<ref id="B43"><label>43</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Venturi</surname> <given-names>V</given-names></name> <name><surname>Price</surname> <given-names>DA</given-names></name> <name><surname>Douek</surname> <given-names>DC</given-names></name> <name><surname>Davenport</surname> <given-names>MP</given-names></name></person-group>. <article-title>The molecular basis for public T-cell responses?</article-title> <source>Nat Rev Immunol</source> (<year>2008</year>) <volume>8</volume>(<issue>3</issue>):<fpage>231</fpage>&#x02013;<lpage>8</lpage>.<pub-id pub-id-type="doi">10.1038/nri2260</pub-id><pub-id pub-id-type="pmid">18301425</pub-id></citation></ref>
<ref id="B44"><label>44</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qi</surname> <given-names>Q</given-names></name> <name><surname>Liu</surname> <given-names>Y</given-names></name> <name><surname>Cheng</surname> <given-names>Y</given-names></name> <name><surname>Glanville</surname> <given-names>J</given-names></name> <name><surname>Zhang</surname> <given-names>D</given-names></name> <name><surname>Lee</surname> <given-names>JY</given-names></name> <etal/></person-group> <article-title>Diversity and clonal selection in the human T-cell repertoire</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2014</year>) <volume>111</volume>(<issue>36</issue>):<fpage>13139</fpage>&#x02013;<lpage>44</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.1409155111</pub-id><pub-id pub-id-type="pmid">25157137</pub-id></citation></ref>
<ref id="B45"><label>45</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Surh</surname> <given-names>CD</given-names></name> <name><surname>Sprent</surname> <given-names>J</given-names></name></person-group>. <article-title>Homeostatic T cell proliferation: how far can T cells be activated to self-ligands?</article-title> <source>J Exp Med</source> (<year>2000</year>) <volume>192</volume>(<issue>4</issue>):<fpage>F9</fpage>&#x02013;<lpage>14</lpage>.<pub-id pub-id-type="doi">10.1084/jem.192.4.F9</pub-id></citation></ref>
<ref id="B46"><label>46</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blachly</surname> <given-names>JS</given-names></name> <name><surname>Ruppert</surname> <given-names>AS</given-names></name> <name><surname>Zhao</surname> <given-names>W</given-names></name> <name><surname>Long</surname> <given-names>S</given-names></name> <name><surname>Flynn</surname> <given-names>J</given-names></name> <name><surname>Flinn</surname> <given-names>I</given-names></name> <etal/></person-group> <article-title>Immunoglobulin transcript sequence and somatic hypermutation computation from unselected RNA-seq reads in chronic lymphocytic leukemia</article-title>. <source>Proc Natl Acad Sci U S A</source> (<year>2015</year>) <volume>112</volume>(<issue>14</issue>):<fpage>4322</fpage>&#x02013;<lpage>7</lpage>.<pub-id pub-id-type="doi">10.1073/pnas.1503587112</pub-id><pub-id pub-id-type="pmid">25787252</pub-id></citation></ref>
<ref id="B47"><label>47</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname> <given-names>SD</given-names></name> <name><surname>Raeburn</surname> <given-names>LA</given-names></name> <name><surname>Holt</surname> <given-names>RA</given-names></name></person-group>. <article-title>Profiling tissue-resident T cell repertoires by RNA sequencing</article-title>. <source>Genome Med</source> (<year>2015</year>) <volume>7</volume>:<fpage>125</fpage>.<pub-id pub-id-type="doi">10.1186/s13073-015-0248-x</pub-id><pub-id pub-id-type="pmid">26620832</pub-id></citation></ref>
<ref id="B48"><label>48</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>B</given-names></name> <name><surname>Li</surname> <given-names>T</given-names></name> <name><surname>Pignon</surname> <given-names>JC</given-names></name> <name><surname>Wang</surname> <given-names>B</given-names></name> <name><surname>Wang</surname> <given-names>J</given-names></name> <name><surname>Shukla</surname> <given-names>SA</given-names></name> <etal/></person-group> <article-title>Landscape of tumor-infiltrating T cell repertoire of human cancers</article-title>. <source>Nat Genet</source> (<year>2016</year>) <volume>48</volume>(<issue>7</issue>):<fpage>725</fpage>&#x02013;<lpage>32</lpage>.<pub-id pub-id-type="doi">10.1038/ng.3581</pub-id><pub-id pub-id-type="pmid">27240091</pub-id></citation></ref>
</ref-list>
</back>
</article>