<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Oncol.</journal-id>
<journal-title>Frontiers in Oncology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Oncol.</abbrev-journal-title>
<issn pub-type="epub">2234-943X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fonc.2025.1625369</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Oncology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Peripheral blood TCR repertoire improves early detection across multiple cancer types utilizing a cancer predictor</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Tang</surname>
<given-names>Yinglei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3060704/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liao</surname>
<given-names>Xinyi</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liao</surname>
<given-names>Bo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Peng</surname>
<given-names>Dejun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Qingbo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Mathematics and Statistics, Hainan Normal University</institution>, <addr-line>Haikou</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>China Unicom (Hainan) Industrial Internet Co. Ltd</institution>, <addr-line>Haikou</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Key Laboratory of Data Science and Intelligence Education, Hainan Normal University, Ministry of Education</institution>, <addr-line>Haikou</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1494920/overview">Tuba Gide</ext-link>, Melanoma Institute Australia, Australia</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/615156/overview">Jiayin Wang</ext-link>, Xi&#x2019;an Jiaotong University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1647830/overview">Xinyang Qian</ext-link>, Xi&#x2019;an Jiaotong University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Bo Liao, <email xlink:href="mailto:boliao@yeah.net">boliao@yeah.net</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>27</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>15</volume>
<elocation-id>1625369</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>29</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Tang, Liao, Liao, Peng and Li.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Tang, Liao, Liao, Peng and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>In the early asymptomatic stages of cancer, the immune system initiates a targeted response against tumor-associated antigens. During this phase, the immune system specifically identifies tumor antigens and triggers the clonal expansion of tumor antigen-specific T cells, which recognize tumor antigen peptides presented by the major histocompatibility complex via the T-cell receptor (TCR) on their surface. Consequently, monitoring alterations in the TCR repertoire holds promise for evaluating an individual&#x2019;s immune status for cancer detection.</p>
</sec>
<sec>
<title>Methods</title>
<p>In this study, we introduced a deep learning framework named DeepCaTCR, designed to enhance the prediction of cancer-associated T-cell receptors. The framework employs a one-dimensional convolutional neural network with variable convolutional kernels, a bidirectional long short-term memory network, and a self-attention mechanism to facilitate feature extraction from amino acid fragments of varying lengths.</p>
</sec>
<sec>
<title>Results</title>
<p>DeepCaTCR demonstrates superior performance in cancer-associated TCR recognition, achieving an area under the receiver operating characteristic curve (AUC) of 0.863 and an F1-score of 0.669, thereby outperforming prevailing deep learning models. Validation result indicates that DeepCaTCR effectively distinguishes between tumor-infiltrating lymphocytes (TILs) and healthy peripheral blood samples, achieving an AUC greater than 0.95. It also exhibits high sensitivity (62.5%) and specificity (over 98%) in peripheral blood testing for early-stage cancer patients. To further enhance detection efficacy, we introduced a variance-based repertoire scoring strategy to quantify the dynamic heterogeneity of TCR clonal amplification, resulting in an increased AUC of 0.967 for pan-cancer early screening.</p>
</sec>
<sec>
<title>Discussion</title>
<p>This study introduces a novel tool for analyzing the tumor immune microenvironment, offering significant translational potential for early cancer diagnosis. Its key feature is a new scoring method based on variance, not the average method.</p>
</sec>
</abstract>
<kwd-group>
<kwd>TCR repertoire</kwd>
<kwd>peripheral blood</kwd>
<kwd>cancer detection</kwd>
<kwd>deep learning</kwd>
<kwd>TCR</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="5"/>
<equation-count count="17"/>
<ref-count count="47"/>
<page-count count="16"/>
<word-count count="8342"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Cancer Immunity and Immunotherapy</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>The high mortality rate of cancer is primarily due to the late-stage diagnosis of many cases, which consequently leads to lost opportunities for early intervention and treatment. Early cancer screening is as crucial for decreasing both the incidence and mortality rates associated with cancer (<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B2">2</xref>). Traditional imaging methods like endoscopy, CT (<xref ref-type="bibr" rid="B3">3</xref>), MRI, and PET (<xref ref-type="bibr" rid="B4">4</xref>) are limited to detecting visible cancerous lesions and face challenges in speed, sensitivity, and effectiveness (<xref ref-type="bibr" rid="B5">5</xref>). Similarly, tumor marker screenings, such as carcinoembryonic and carbohydrate antigen tests (<xref ref-type="bibr" rid="B6">6</xref>), are practical but lack specificity due to the absence of unique markers for many cancer types. Advancements in Artificial Intelligence (AI) have enhanced early cancer screening by creating diagnostic models using tumor marker concentrations (<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B8">8</xref>). Circulating free DNA is a key tool in cancer detection (<xref ref-type="bibr" rid="B9">9</xref>), but its plasma concentration can be obscured by noise, complicating early cancer detection. Additionally, the immune system&#x2019;s response to early-stage cancers produces immune characteristics that, when combined with AI, could serve as immune biomarkers for intelligent early screening models (<xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B11">11</xref>).</p>
<p>The tumor microenvironment (TME) is vital in influencing the immune response to cancer by modulating T-cell activity (<xref ref-type="bibr" rid="B12">12</xref>). Antigen-specific T cells in the TME are crucial for identifying and attacking tumor antigens (<xref ref-type="bibr" rid="B13">13</xref>), aided by the diverse and adaptable T-cell receptor (TCR) repertoire. This diversity is key for effectively targeting cancer cells (<xref ref-type="bibr" rid="B14">14</xref>). The expansion and diversification of the TCR repertoire enable T cells to recognize tumor antigens and activate them. Analyzing the TCR repertoire is a powerful approach to understanding the clonal responses of tumor-reactive T cells (<xref ref-type="bibr" rid="B15">15</xref>), which are crucial for effective antitumor immune responses. The TCR repertoire provides a detailed map of the diversity and specificity of T cells, which can be used to track the dynamics of immune responses in cancer. Recent advancements in sequencing technologies have enabled the comprehensive analysis of TCR repertoires (<xref ref-type="bibr" rid="B16">16</xref>), allowing researchers to identify specific T-cell clones that are reactive to tumor antigens and to understand their role in the immune response against cancer (<xref ref-type="bibr" rid="B17">17</xref>). A study demonstrated that the oligoclonal expansion of TCR &#x3b2; clonotypes is associated with effective immune checkpoint therapy responses, suggesting that specific TCR signatures can serve as biomarkers for predicting treatment outcomes (<xref ref-type="bibr" rid="B18">18</xref>).</p>
<p>Numerous computational approaches have been devised to detect cancer-associated sequences and estimate cancer probability. However, the identification of cancer-associated T-cell receptors (caTCRs) through computational methods encounters three primary challenges: 1) the presence of weak immune signals attributable to the low neoantigen burden characteristic of early-stage tumors, 2) the conservation of TCR motifs across various cancer types, and 3) the sparse distribution of informative TCR sequences. Although current methodologies offer partial solutions to these challenges, they continue to exhibit significant limitations. Beshnova et&#xa0;al. used convolutional neural networks to differentiate cancer TCRs but covered limited data (<xref ref-type="bibr" rid="B19">19</xref>). Xu et&#xa0;al. (<xref ref-type="bibr" rid="B20">20</xref>) and Qian et&#xa0;al. (<xref ref-type="bibr" rid="B21">21</xref>) employed an enhanced TextCNN network with 1-max pooling and manual filter allocation to identify breast cancer and lung cancer, which may result in the loss of key long-range motifs. Zhang et&#xa0;al. used a pre-trained protein language model to capture TCR sequence features, but its early cancer detection sensitivity is limited by training data bias (<xref ref-type="bibr" rid="B22">22</xref>). Cai et&#xa0;al. showed good performance in pan-cancer screening but struggled with early immune microenvironment features (<xref ref-type="bibr" rid="B23">23</xref>).</p>
<p>To overcome these challenges, we proposed DeepCaTCR, a deep learning framework that integrates three key innovations. First, we employed multi-scale k-max pooling to capture variable-length motifs (two to five amino acids) while preserving the top k informative segments per filter. Unlike 1-max pooling (in DeepLION, DeepLION2, and BertTCR), this approach mitigates bias toward dominant but non-specific signals and enhances sensitivity to sparse caTCR features. Second, we introduced context-aware feature fusion via bidirectional long short-term memory (LSTM) (BiLSTM) layers, modeling dependencies between discontinuous TCR segments to address motif conservation variability. Third, we implemented a noise-resistant attention mechanism [multi-head self-attention (MHSA)] after k-max pooling to dynamically weight informative sequence regions, suppressing noise from non-cancerous motifs. Our approach uniquely combines these components to enhance caTCR detection in early-stage tumors.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<p>In this study, we developed the deep learning framework DeepCaTCR, which effectively manages the varying lengths of amino acid fragments in TCR sequences. Initially, we <italic>de novo</italic> assembled cancer-associated TCRs from RNA-seq data and collected non-cancer TCRs from healthy individuals to create a training dataset. Subsequently, we constructed a pattern recognition network utilizing deep learning algorithms to extract features from amino acid fragments of differing lengths. Finally, we implemented a variance repertoire scoring strategy to quantify individual cancer scores. This study differentiates between cancerous and healthy individuals based on TCR repertoire derived from TCR-seq, exploring non-invasive early cancer detection methods.</p>
<sec id="s2_1">
<label>2.1</label>
<title>Datasets</title>
<sec id="s2_1_1">
<label>2.1.1</label>
<title>TCR training data and data processing</title>
<p>The positive training data were generated from CDR3s identified by TRUST (<xref ref-type="bibr" rid="B24">24</xref>) from The Cancer Genome Atlas (TCGA) 4,200 tumor RNA-seq samples across 32 cancer types (<xref ref-type="bibr" rid="B25">25</xref>). Detailed information on the specific samples is available in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;1</bold>
</xref>. This approach was chosen instead of utilizing TCR repertoires from tumor or blood cancer sources. These <italic>de novo</italic> assembled caTCRs from RNA-seq data showed higher specificity than those from TCR-seq data. Only the TCR &#x3b2; chain CDR3 region, crucial for antigenic specificity, was used. TRUST-assembled CDR3 sequences excluded incomplete sequences (not starting with C and ending with F), non-productive sequences (containing B and *), those common in healthy individuals, and sequences shorter than 10 or longer than 24. Negative data from the training set were derived from TCR-seq data of healthy individuals&#x2019; peripheral blood (<xref ref-type="bibr" rid="B26">26</xref>) by selecting CDR3 sequences with clonal frequencies at least four times the minimum in each TCR repertoire and clustering them using iSMART (<xref ref-type="bibr" rid="B27">27</xref>). Incomplete, unproductive, and improperly sized sequences were excluded. This process yielded 30,000 cancer-associated and 59,851 normal CDR3 sequences, mostly ranging from 11 to 20 in length. In this study, only sequences of length 11 to 20 were used for training and validation.</p>
</sec>
<sec id="s2_1_2">
<label>2.1.2</label>
<title>TCR repertoire data and data processing</title>
<p>The TCR cohort repertoire data utilized in this study were obtained from bulk TCR sequencing. The cancer tumor-infiltrating lymphocyte (TIL) cohort comprises samples from breast cancer (BRCA) (<xref ref-type="bibr" rid="B28">28</xref>), lung metastasis (Lung BM) (<xref ref-type="bibr" rid="B29">29</xref>), lung cancer (<xref ref-type="bibr" rid="B29">29</xref>), melanoma (MELA) (<xref ref-type="bibr" rid="B30">30</xref>), and pancreatic cancer (PC) (<xref ref-type="bibr" rid="B31">31</xref>). The cancer peripheral blood mononuclear cell (PBMC) cohort includes samples from BRCA (<xref ref-type="bibr" rid="B28">28</xref>), MELA (<xref ref-type="bibr" rid="B32">32</xref>), ovarian cancer (OV) (<xref ref-type="bibr" rid="B33">33</xref>), PC (<xref ref-type="bibr" rid="B31">31</xref>), colorectal cancer (CRC) (<xref ref-type="bibr" rid="B34">34</xref>), bladder cancer (<xref ref-type="bibr" rid="B35">35</xref>), glioblastoma multiforme (GBM) (<xref ref-type="bibr" rid="B36">36</xref>), and lung cancer (<xref ref-type="bibr" rid="B37">37</xref>). The cancer staging PBMC cohort encompasses stage I&#x2013;II lung cancer (<xref ref-type="bibr" rid="B38">38</xref>), stage III lung cancer (<xref ref-type="bibr" rid="B38">38</xref>), stage I renal cell carcinoma (RCC) (<xref ref-type="bibr" rid="B19">19</xref>), borderline ovarian cancer (<xref ref-type="bibr" rid="B19">19</xref>), stage II&#x2013;III ovarian cancer (<xref ref-type="bibr" rid="B19">19</xref>), and stage II PC (<xref ref-type="bibr" rid="B19">19</xref>). The non-cancer PBMC cohorts consist of samples from yellow fever virus (YFV) (<xref ref-type="bibr" rid="B39">39</xref>), human cytomegalovirus (HCMV) (<xref ref-type="bibr" rid="B26">26</xref>), healthy T-cell controls (Healthy TC) (<xref ref-type="bibr" rid="B40">40</xref>), graft-versus-host disease (GVHD) (<xref ref-type="bibr" rid="B41">41</xref>), healthy donors (HCMV&#x2212;) (<xref ref-type="bibr" rid="B26">26</xref>), and healthy donors (<xref ref-type="bibr" rid="B42">42</xref>). The details of the datasets are provided in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;1</bold>
</xref>. In the preprocessing of repertoire data, TCR sequences with lengths ranging from 11 to 20 nucleotides were selected. Following the exclusion of unqualified TCR sequences as detailed in Section 3.1, the sequences with the top 10,000 clone scores were retained for further analysis. These sequences were subsequently clustered using the iSMART algorithm. The TCR sequences resulting from this clustering process were considered in this study to be those most likely associated with cancer.</p>
</sec>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Multi-scale attentive BiLSTM for TCR motif analysis</title>
<p>
<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> presents a structural diagram of the TCR sequence recognition algorithm. In summary, TCR sequences associated with cancer and those not associated with cancer are initially encoded into a matrix using amino acid biochemical features as model inputs. This matrix is subsequently processed in the convolutional layer using a multi-scale convolutional kernel to extract features. A max pooling layer is employed to encode the feature set of amino acid fragments of varying lengths before applying a multi-head self-attention mechanism to assign differential attentional weights. The resulting attention-weighted encoding matrices are interconnected along the channel dimension, producing an attention-weighted matrix that contains key molecules of different lengths. This weighted pattern matrix is then further processed using bidirectional long- and short-term memory networks, which focus on the correlations between these key patterns. Finally, a self-attention mechanism is introduced to assign varying attention weights, followed by the application of a linear classifier for binary classification.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Structure diagram of TCR sequence recognition algorithm. TCR, T-cell receptor.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1625369-g001.tif">
<alt-text content-type="machine-generated">Flowchart of a deep learning model for predicting cancer probability using TCR Matrix. It shows convolution, ReLU activation, 3-max pooling, and multi-head self-attention. The process creates feature maps and an attention matrix for each channel, which are concatenated into a multi-scale attention matrix. This matrix is processed with a bidirectional Long Short-Term Memory (LSTM) and soft attention, leading to an output vector representing cancer probability.</alt-text>
</graphic>
</fig>
<sec id="s2_2_1">
<label>2.2.1</label>
<title>1D convolutional neural network</title>
<p>Deep convolutional neural networks (CNNs) are a class of deep learning algorithms adept at identifying latent patterns within grid data. CNNs serve as highly effective tools for feature extraction from such data, often outperforming traditional machine learning algorithms (<xref ref-type="bibr" rid="B23">23</xref>). However, when CNNs are employed to extract features from equal-length sequence encoding matrices, created by padding variable-length sequences with zero vectors, the model performance tends to degrade. This degradation is likely due to the introduction of zero vectors via AA index encoding, which alters the original data length distribution and introduces significant noise. To mitigate this issue, we used a one-dimensional CNN (1D CNN) algorithm to transform the encoding matrix into a one-dimensional sequence. This approach more effectively preserves sequence information and the dependencies between sequences (<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B21">21</xref>). Let the input sequence be represented as a matrix <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <italic>L</italic> is the padded sequence length and <italic>d</italic> is the encoding dimension (amino acid index features). The 1D convolution operation applies a filter <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> with kernel size <italic>k</italic>, sliding over the sequence to generate feature maps (<xref ref-type="disp-formula" rid="eq1">Equation&#xa0;1</xref>):</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the subsequence window from position <italic>i</italic> to <italic>i + k</italic> &#x2212; 1, and <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a bias term.</p>
</sec>
<sec id="s2_2_2">
<label>2.2.2</label>
<title>k-max pooling</title>
<p>Furthermore, we employed the <italic>k</italic>-max pooling algorithm to transform the one-dimensional coding sequence into a sequence of uniform length, effectively mitigating interference from zero vector padding, and <italic>k</italic>-max pooling selects the k largest values from the feature map <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="disp-formula" rid="eq2">Equation&#xa0;2</xref>):</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> retains the <italic>k</italic> highest activations.</p>
</sec>
<sec id="s2_2_3">
<label>2.2.3</label>
<title>Multi-scale convolutional kernels</title>
<p>The currently employed algorithm is limited to acquiring amino acid fragments of a fixed length from the sequence. However, prior research has demonstrated that the length of cancer-related key motifs is variable, typically ranging from two to eight amino acids. To capture the characteristics of amino acid fragments of varying lengths, this study adapted the TextCNN model from natural language processing, implementing convolutional kernels of diverse sizes within the convolutional layer (<xref ref-type="bibr" rid="B20">20</xref>). To capture motifs of variable lengths (<italic>k</italic>
<sub>1</sub>, <italic>k</italic>
<sub>2</sub>, &#x2026;, <italic>k</italic>
<sub>n</sub>), parallel convolutional kernels of different sizes are applied (<xref ref-type="disp-formula" rid="eq3">Equation&#xa0;3</xref>):</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mo>&#x2295;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the feature map from the <italic>j</italic>th kernel and &#x2a01; denotes concatenation along the channel dimension.</p>
</sec>
<sec id="s2_2_4">
<label>2.2.4</label>
<title>Self-attention mechanism</title>
<p>In the context of the weighted motif matrix of a sequence, it is acknowledged that amino acid fragments of varying lengths exert differential influences on sequence specificity. To address this, a self-attention mechanism was implemented to evaluate the similarity between different positions within the sequence, assigning an attention weight to each position. This allows the model to autonomously identify the key motifs within the sequence. Given the multi-scale feature matrix <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the attention weights <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> for each position <italic>i</italic> are computed as follows (<xref ref-type="disp-formula" rid="eq4">Equation&#xa0;4</xref>):</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:msup>
<mml:mi>K</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>K</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>K</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are learnable query/key matrices. The attention-weighted output is as follows (<xref ref-type="disp-formula" rid="eq5">Equation&#xa0;5</xref>):</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>L</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced>
<mml:mi>i</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s2_2_5">
<label>2.2.5</label>
<title>Bidirectional long short-term memory</title>
<p>Nonetheless, it has been observed that this algorithmic approach neglects the interconnections between key motifs within the same sequence. LSTM networks, a class of neural networks specifically designed for sequential data processing, offer a potential solution. In LSTM networks, the output at each time step, known as the hidden state, encapsulates all input information up to that point. Additionally, the cell state serves as a repository for long-term information. The input gate computes an activation value based on the current input and the state from the preceding moment to determine the acceptance of new input. Similarly, the forgetting gate calculates the degree of forgetting by evaluating the current input alongside the previous state. Activation values for each gate are computed based on the hidden state from the preceding moment.</p>
<p>In contrast, the BiLSTM model processes sequential data by considering not only the current position at each time step but also both preceding positions (via the forward LSTM) and subsequent positions (via the backward LSTM). This dual processing results in the generation of two hidden states at each time step: one derived from the forward network and the other from the backward network. These hidden states are subsequently combined to form a comprehensive context representation that encapsulates enduring dependency information within the text. Consequently, this model is capable of capturing more profound contextual associations. BiLSTM processes the attention-weighted matrix <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to model long-range dependencies (<xref ref-type="bibr" rid="B44">44</xref>). For each time step <italic>t</italic>, the forward (<inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>h</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) and backward (<inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>h</mml:mi>
<mml:mo>&#x2190;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) hidden states are computed as follows (<xref ref-type="disp-formula" rid="eq6">Equation&#xa0;6</xref>):</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>h</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>M</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>h</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>h</mml:mi>
<mml:mo>&#x2190;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>M</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>h</mml:mi>
<mml:mo>&#x2190;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The final hidden state combines both directions (<xref ref-type="disp-formula" rid="eq7">Equation&#xa0;7</xref>):</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfenced close="]" open="[">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>h</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2225;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>h</mml:mi>
<mml:mo>&#x2190;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where || denotes concatenation.</p>
</sec>
<sec id="s2_2_6">
<label>2.2.6</label>
<title>Classification layer</title>
<p>Prior to the introduction of the attention mechanism, we input the weighted motif matrix into BiLSTM to evaluate the correlation between different key motifs of the sequence, thereby adaptively capturing the long-range dependencies between amino acid fragments. The aggregated hidden states <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are fed into a fully connected layer with softmax for binary classification (<xref ref-type="disp-formula" rid="eq8">Equation&#xa0;8</xref>):</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im19">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are learnable parameters.</p>
</sec>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Cancer predictor</title>
<sec id="s2_3_1">
<label>2.3.1</label>
<title>TCR repertoire mean scoring strategy</title>
<p>Let <italic>R</italic> = {TCR<sub>1</sub>, TCR<sub>2</sub>, &#x2026;, TCR<italic>
<sub>N</sub>
</italic>} represent a TCR repertoire containing <italic>N</italic> distinct TCRs, and the composite score of the TCR repertoire <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mfenced>
<mml:mi>R</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is defined as the arithmetic mean of the predicted cancer scores across all TCRs in <italic>R</italic> (<xref ref-type="disp-formula" rid="eq9">Equation&#xa0;9</xref>):</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mfenced>
<mml:mi>R</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> denotes the predicted cancer score of the <italic>i</italic>th TCR (<italic>i</italic> = 1, 2, &#x2026;, <italic>N</italic>). This formulation reflects the intuition that the overall repertoire score represents the average likelihood of cancer-associated specificity across its constituent TCRs.</p>
</sec>
<sec id="s2_3_2">
<label>2.3.2</label>
<title>TCR repertoire variance scoring strategy</title>
<p>Let <italic>R</italic> = {TCR<sub>1</sub>, TCR<sub>2</sub>, &#x2026;, TCR<italic>
<sub>N</sub>
</italic>} represent a TCR repertoire containing <italic>N</italic> distinct TCRs, and the variance-based composite score <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mfenced>
<mml:mi>R</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is then defined as the variance of the predicted cancer scores across all TCRs in <italic>R</italic> (<xref ref-type="disp-formula" rid="eq10">Equation&#xa0;10</xref>):</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mfenced>
<mml:mi>R</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mfenced>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mfenced>
<mml:mi>R</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
<mml:mfenced>
<mml:mi>R</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is the mean predicted cancer score (as defined in the mean strategy). This formulation quantifies the spread (heterogeneity) of predicted cancer scores within the repertoire, with higher variance indicating greater diversity in cancer-associated specificity among TCRs.</p>
</sec>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>TCR sequence recognition model parameter settings</title>
<p>The model architecture and final hyperparameter configuration, including convolutional kernel dimensions, pooling strategies, and fully connected layer specifications, are detailed in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The process of parameter tuning, which involves a systematic evaluation of alternative dropout rates and learning rates, along with the associated performance metrics, is thoroughly documented in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;2</bold>
</xref>.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>TCR sequence recognition model architecture and hyperparameters.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Layer/component</th>
<th valign="middle" align="left">Parameter setting</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Input encoding</td>
<td valign="middle" align="left">1. TCR sequence encoded as L &#xd7; 15 matrix.<break/>2. Zero-padded to 20 &#xd7; 15 if L &lt; 20.</td>
</tr>
<tr>
<td valign="middle" align="left">Multi-scale convolution</td>
<td valign="middle" align="left">1. Kernel widths: fixed at 15 (matches input dimension).<break/>2. Kernel heights: 2, 3, 4, 5.<break/>3. Kernels per height: 4 (total 16 kernels).</td>
</tr>
<tr>
<td valign="middle" align="left">Max pooling</td>
<td valign="middle" align="left">1. Window size: 3.<break/>2. Output: 3 &#xd7; 4 matrix <italic>P</italic>.</td>
</tr>
<tr>
<td valign="middle" align="left">Multi-head self-attention</td>
<td valign="middle" align="left">1. Attention heads: 2.<break/>2. Hidden dimension: 4 (aligned with P).<break/>3. Subspace projection for <italic>Q</italic>, <italic>K</italic>, <italic>V</italic>.</td>
</tr>
<tr>
<td valign="middle" align="left">Bidirectional LSTM</td>
<td valign="middle" align="left">1. Input dimension: 3.<break/>2. Hidden dimension: context-aware (self-attention adjusted).<break/>3. Output: concatenated forward/backward states.</td>
</tr>
<tr>
<td valign="middle" align="left">Fully connected layer</td>
<td valign="middle" align="left">1. Units: 6.<break/>2. Dropout: 50% regularization.<break/>3. Activation: softmax (binary classification).</td>
</tr>
<tr>
<td valign="middle" align="left">Output</td>
<td valign="middle" align="left">Probabilities for cancer/non-cancer classes.</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>LSTM, long short-term memory.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Model training and evaluation</title>
<p>The experiments were executed on a high-performance computing platform operating Ubuntu 20.04, featuring an Intel<sup>&#xae;</sup> Xeon<sup>&#xae;</sup> Platinum 8470Q processor with 20 virtual CPUs, 90GB of RAM, and an NVIDIA virtual GPU with 48GB of memory. The software environment consisted of Python 3.8 and PyTorch 1.10.0 with CUDA 11.3 for acceleration, supplemented by standard scientific computing libraries. For model development, 30,000 cancer-associated CDR3 sequences and approximately 60,000 non-cancer sequences were encoded, assigning binary labels (1 for cancer and 0 for non-cancer). The dataset was divided using stratified sampling, with 80% designated for training and 20% for validation. To ensure robust performance evaluation, fivefold cross-validation was employed across all experiments. The training process utilized the Adam optimizer with a learning rate of 0.001 and cross-entropy loss for error computation. To mitigate overfitting, dropout was applied with a probability of 0.5 during training. The model was trained for a maximum of 1,000 epochs, with an early stopping criterion activated if the validation loss did not improve for 20 consecutive epochs.</p>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Validation metrics</title>
<p>This study utilized six metrics to assess model performance: accuracy (ACC), sensitivity (SEN), specificity (SPE), area under the receiver operating characteristic curve (AUC), F1-score, and Matthews Correlation Coefficient (MCC). Each metric offers unique insights into the classifier&#x2019;s capabilities (<xref ref-type="disp-formula" rid="eq11">Equations&#xa0;11</xref>&#x2013;<xref ref-type="disp-formula" rid="eq17">17</xref>):</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq15">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq16">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq17">
<label>(17)</label>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mfenced>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mfenced>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mfenced>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mfenced>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>TP</italic>, <italic>TN</italic>, <italic>FP</italic>, and <italic>FN</italic> represent true-positive, true-negative, false-positive, and false-negative predictions, respectively.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<sec id="s3_1">
<label>3.1</label>
<title>Model performance in recognizing caTCRs</title>
<p>Due to the inability to directly utilize raw amino acid sequences for model training, this study employed a biochemical feature-based encoding strategy to convert these sequences into numerical form. Focusing on the functional characteristics of antigen-binding sites within the CDR3 region, 553 biochemical feature indicators of amino acids were selected from the AAindex database for principal component analysis (PCA). Through dimensionality reduction, a 20 &#xd7; 20 amino acid feature matrix was derived, and the top 15 principal components, which collectively accounted for over 95% of the cumulative variance, were chosen to construct a standardized 20 &#xd7; 15 AAindex coding matrix. For CDR3 sequences shorter than 20 amino acids, a zero-padding strategy was applied to encode them uniformly into a 20 &#xd7; 15 matrix structure.</p>
<p>In order to assess the performance of DeepCaTCR, we conducted a comparative analysis with leading caTCR recognition models, namely, DeepLION and BertTCR. This evaluation utilized a consistent encoding scheme, training dataset, learning rate, and batch size across all models. Through fivefold cross-validation, DeepCaTCR demonstrated superior performance in antigen-specific TCR recognition, achieving an ACC of 0.807 &#xb1; 0.003 and AUC of 0.863 &#xb1; 0.003, as presented in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. Notably, DeepCaTCR outperformed both DeepLION (ACC: 0.801, AUC: 0.854) and BertTCR (ACC: 0.760, AUC: 0.790), achieving the highest ACC and AUC values. The sensitivity of DeepCaTCR (0.586) was 28% higher than that of BertTCR (0.457), while its specificity (0.918) remained the highest among all models evaluated. Furthermore, the F1-score (0.669) and MCC (0.548) exceeded those of the competing models.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Performance comparison of caTCR recognition models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Model</th>
<th valign="middle" align="left">ACC</th>
<th valign="middle" align="left">AUC</th>
<th valign="middle" align="left">SEN</th>
<th valign="middle" align="left">SPE</th>
<th valign="middle" align="left">F1</th>
<th valign="middle" align="left">MCC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">DeepCaTCR</td>
<td valign="middle" align="left">0.807 &#xb1; 0.003</td>
<td valign="middle" align="left">0.863 &#xb1; 0.003</td>
<td valign="middle" align="left">0.586 &#xb1; 0.026</td>
<td valign="middle" align="left">0.918 &#xb1; 0.010</td>
<td valign="middle" align="left">0.669 &#xb1; 0.013</td>
<td valign="middle" align="left">0.548 &#xb1; 0.009</td>
</tr>
<tr>
<td valign="middle" align="left">DeepLION</td>
<td valign="middle" align="left">0.801 &#xb1; 0.003</td>
<td valign="middle" align="left">0.854 &#xb1; 0.004</td>
<td valign="middle" align="left">0.577 &#xb1; 0.014</td>
<td valign="middle" align="left">0.913 &#xb1; 0.009</td>
<td valign="middle" align="left">0.659 &#xb1; 0.006</td>
<td valign="middle" align="left">0.533 &#xb1; 0.007</td>
</tr>
<tr>
<td valign="middle" align="left">BertTCR</td>
<td valign="middle" align="left">0.760 &#xb1; 0.003</td>
<td valign="middle" align="left">0.790 &#xb1; 0.004</td>
<td valign="middle" align="left">0.457 &#xb1; 0.022</td>
<td valign="middle" align="left">0.911 &#xb1; 0.009</td>
<td valign="middle" align="left">0.559 &#xb1; 0.017</td>
<td valign="middle" align="left">0.425 &#xb1; 0.009</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR-noBiLSTM</td>
<td valign="middle" align="left">0.795 &#xb1; 0.006</td>
<td valign="middle" align="left">0.848 &#xb1; 0.005</td>
<td valign="middle" align="left">0.593 &#xb1; 0.032</td>
<td valign="middle" align="left">0.895 &#xb1; 0.024</td>
<td valign="middle" align="left">0.657 &#xb1; 0.007</td>
<td valign="middle" align="left">0.520 &#xb1; 0.008</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR-noMHSA</td>
<td valign="middle" align="left">0.800 &#xb1; 0.003</td>
<td valign="middle" align="left">0.853 &#xb1; 0.005</td>
<td valign="middle" align="left">0.602 &#xb1; 0.013</td>
<td valign="middle" align="left">0.898 &#xb1; 0.011</td>
<td valign="middle" align="left">0.666 &#xb1; 0.003</td>
<td valign="middle" align="left">0.532 &#xb1; 0.007</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR-noBiLSTM-noMHSA</td>
<td valign="middle" align="left">0.791 &#xb1; 0.005</td>
<td valign="middle" align="left">0.840 &#xb1; 0.006</td>
<td valign="middle" align="left">0.560 &#xb1; 0.020</td>
<td valign="middle" align="left">0.906 &#xb1; 0.014</td>
<td valign="middle" align="left">0.641 &#xb1; 0.008</td>
<td valign="middle" align="left">0.509 &#xb1; 0.009</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>ACC, accuracy; AUC, area under the receiver operating characteristic curve; SEN, sensitivity; SPE, specificity; MCC, Matthews Correlation Coefficient.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>We conducted a detailed analysis of the performance metrics for each fold and performed paired t-tests to assess statistical significance, comparing each model against DeepCaTCR. The findings indicated that BertTCR was significantly outperformed by DeepCaTCR across all metrics (p &lt; 0.0001), with the exception of specificity (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). DeepLION demonstrated significantly lower performance than DeepCaTCR in terms of ACC, AUC, and MCC (p = 0.01&#x2013;0.02). We posited that the suboptimal performance of BertTCR could be attributed to its limited number of filters, which was initially set at six. To investigate this hypothesis, we increased the number of filters in BertTCR to nine, resulting in a significant enhancement in model performance (p &lt; 0.008, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Comparison of model performance using fivefold cross-validation. <bold>(A&#x2013;F)</bold> The results for six evaluation metrics: <bold>(A)</bold> ACC, <bold>(B)</bold> AUC, <bold>(C)</bold> SEN, <bold>(D)</bold> SPE, <bold>(E)</bold> F1-score, and <bold>(F)</bold> MCC. The models under comparison include DeepCaTCR, DeepCaTCR-noBiLSTM, DeepCaTCR-noMHSA, DeepCaTCR-noBiLSTM-noMHSA, DeepLION, and BertTCR. The box plots depict the distribution of each metric across the five folds, while individual data points indicate the metric value for each fold. Statistical significance was evaluated using paired t-tests, with each model compared against DeepCaTCR as the reference model. TCR, T-cell receptor; ACC, accuracy; AUC, area under the receiver operating characteristic curve; SEN, sensitivity; SPE, specificity; MCC, Matthews Correlation Coefficient.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1625369-g002.tif">
<alt-text content-type="machine-generated">Box plots comparing various models labeled A to F for different metrics: ACC, AUC, SEN, SPE, F1-score, and MCC, with statistical significance indicated by p-values. Each plot presents several models including BertTCR, DeepION, and variations of DeepCatTCR, showing differences in performance for each metric.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Functional analysis of DeepCaTCR key modules</title>
<p>To further substantiate the contributions of the core components within DeepCaTCR, we performed ablation studies focusing on BiLSTM and MHSA. We developed variant models, namely, DeepCaTCR-noBiLSTM, DeepCaTCR-noMHSA, and DeepCaTCR-noBiLSTM-noMHSA. We subjected these ablation variants to the same experimental conditions as the baseline DeepCaTCR model, encompassing input data preprocessing, shared embedding layer parameters, output layer architecture, loss function, optimizer configuration, and train/test splits. The sole alteration involved the exclusion of specific model components.</p>
<p>The comprehensive DeepCaTCR model demonstrated superior performance across several metrics, achieving the highest accuracy (0.807), AUC (0.863), specificity (0.918), F1-score (0.669), and MCC (0.548). This underscores the synergistic advantages of integrating BiLSTM and MHSA. Upon the exclusion of BiLSTM (DeepCaTCR-noBiLSTM), there were notable declines in performance metrics: ACC decreased by 1.5%, AUC by 1.7%, SPE by 2.5%, F1 by 1.8%, and MCC by 5.1% (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). Interestingly, SEN exhibited a slight improvement (0.593 compared to 0.586), which may be attributed to the reduced complexity of the model influencing class-specific predictions. The removal of MHSA (DeepCaTCR-noMHSA) resulted in smaller yet consistent reductions in ACC (0.9%), AUC (1.2%), SPE (2.2%), F1 (0.4%), and MCC (2.9%). Similar to the removal of BiLSTM, SEN improved (0.602 compared to 0.586), suggesting that attention mechanisms may trade off some sensitivity for specificity. The most pronounced degradation in performance metrics (ACC: &#x2212;2.0%, AUC: &#x2212;2.7%, F1: &#x2212;4.2%, MCC: &#x2212;7.1%) underscores the complementary roles of BiLSTM and MHSA in feature extraction and context modeling. Notably, SEN experienced a sharp decline (0.560 compared to 0.586), indicating that the combined use of BiLSTM and MHSA enhances recall for positive samples.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Model performance in cancer patient identification</title>
<p>While DeepCaTCR exhibits strong capabilities in recognizing cancer-associated sequences, its effectiveness in clinically distinguishing between cancer patients and healthy individuals requires further validation through independent experiments. It is important to note that accurately evaluating the overall immune status presents substantial technical challenges. This difficulty arises because antigen-specific TCRs constitute only a small fraction of an individual&#x2019;s TCR repertoire, typically less than 0.1%, and there is a considerable background noise (<xref ref-type="bibr" rid="B43">43</xref>). To address this issue, the study employed the iSMART antigen-specific clustering technique to extract representative sequences from each database. This approach enabled the quantification of an individual&#x2019;s tumor immune response by calculating the mean cancer probability, or cancer score, of these characteristic sequences.</p>
<p>DeepCaTCR demonstrated robust discriminatory power across diverse sample types and clinical scenarios, as quantified by the mean cancer score of antigen-specific TCR clusters. TILs exhibited significantly higher cancer scores than PBMCs from healthy donors (p &lt; 5e&#x2212;07, Wilcoxon rank-sum test, <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>). There was near-perfect discrimination (AUC &gt; 0.95) for all cancer types (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>), with primary lung cancer (AUC = 1), pancreatic cancer (AUC = 0.998), and melanoma (AUC = 0.994) showing high specificity (SPE &gt; 0.96) and sensitivity (SEN = 1.0). Untreated cancer patients had significantly higher PBMC cancer scores than healthy controls (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3C</bold>
</xref>, p &lt; 0.0007), with ovarian cancer (AUC = 0.997) and pancreatic cancer (AUC = 0.989) having the leading performance (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3D</bold>
</xref>). Treated patients (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3E</bold>
</xref>) displayed reduced cancer scores versus untreated cohorts, likely due to the therapy-induced depletion of tumor-reactive T cells. Despite lower scores, the most model-maintained AUC &gt; 0.81 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>) was for refractory cancers (glioblastoma: AUC = 0.814; bladder cancer: AUC = 0.83; CRC: AUC = 0.919), although lung cancer discrimination declined (AUC = 0.667), potentially reflecting prolonged T-cell exhaustion.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>The predictive performance was evaluated by DeepCaTCR utilizing the mean scorer. <bold>(A, C, E)</bold> Box plots and scatter plots illustrating the distribution of cancer scores across various scenarios: <bold>(A)</bold> tumor-infiltrating lymphocytes (TILs) from different cancer types, <bold>(C)</bold> peripheral blood from untreated cancer patients, and <bold>(E)</bold> peripheral blood from treated cancer patients. The sample sizes are indicated on the x-axis. Comparisons were conducted with a cohort of healthy donors (n = 176), and statistical significance was assessed using the Wilcoxon rank-sum test. The red solid line denotes the average predicted score for each donor on the y-axis. <bold>(B, D)</bold> ROC curves and AUC values for cancer patients, using healthy donors (n = 176) as the control group. <bold>(B, D)</bold> The model&#x2019;s performance in predicting TIL samples and untreated PBMC samples, respectively. <bold>(F)</bold> Box plots and scatter plots that depict the distribution of cancer scores from various virus-infected or healthy donors. Comparisons were made with the untreated cancer cohort (n = 137), employing the same statistical significance assessment method as in panel <bold>(A)</bold>. <bold>(G)</bold> ROC curves and AUC values are presented for distinguishing between different virus-infected and healthy donors, using the untreated cancer cohort (n = 137) as the control group. <bold>(H)</bold> A scatter plot illustrates the association between age (x-axis) and cancer risk score (y-axis) within a cohort of healthy participants (n = 176). LOWESS smooth curve was added on top of the scatter plot to display the trend of change. Spearman&#x2019;s rank correlation analysis was conducted, with the correlation coefficient (R) and statistical significance presented in the plot inset. <bold>(I)</bold> Comparative analysis of cancer scores between male and female healthy individuals. The p-value derived from the Wilcoxon rank-sum test is indicated on the plot. ROC, receiver operating characteristic; AUC, area under the receiver operating characteristic curve; PBMC, peripheral blood mononuclear cell.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1625369-g003.tif">
<alt-text content-type="machine-generated">Box plots, ROC curves, and scatter plot analyze cancer scores across various groups. Key comparisons include different cancer types, stages, immune conditions, and control donors. P-values and ROC curves are provided for sensitivity and specificity, and the scatter plot examines cancer scores by age, with no significant correlation identified.</alt-text>
</graphic>
</fig>
<p>Notably, DeepCaTCR maintained high specificity in non-cancer contexts. Evaluation of virus-infected and healthy cohorts (<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3F, G</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;4</bold>
</xref>) revealed consistently strong performance metrics: YFV (AUC = 0.992, SPE = 1.0), GVHD (AUC = 0.984, SPE = 0.933), and HCMV (AUC = 0.956, SPE = 0.899). Healthy donors (AUC = 0.978, SPE = 0.955) and additional healthy samples (AUC = 0.947, SPE = 0.89) further confirmed the model&#x2019;s ability to distinguish cancer-associated TCRs from benign immune responses. Preliminary observations suggested that, despite several elevated cancer scores, the scores within the healthy cohort remain relatively low. Subsequent validation using additional cohorts comprising both healthy and virus-infected individuals demonstrated that the cancer scores fall within anticipated ranges (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3F</bold>
</xref>), indicating that the initial findings may be attributable to the characteristics of the study population rather than methodological flaws.</p>
<p>Further analysis of the correlation between cancer scores and demographic variables such as age and gender yielded a Spearman&#x2019;s correlation coefficient of R = 0.033 (p = 0.66, <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3H</bold>
</xref>), while a comparison by gender using the Wilcoxon test resulted in a p-value of 0.58 (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3I</bold>
</xref>), indicating no significant association. Additionally, we examined the correlation between cancer scores and TCR counts. In the initial healthy cohort (n = 176), a marginal correlation was observed (R = 0.15, p = 0.044, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2A</bold>
</xref>). In the subsequent validation cohort (n = 82), a significant but stronger correlation was found (R = &#x2212;0.5, p = 2.1e&#x2212;06, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2B</bold>
</xref>). However, in the combined analysis (n = 258), no correlation was detected (R = 0.021, p = 0.74, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2C</bold>
</xref>). The results suggest that cancer scores are generally stable within healthy populations, unaffected by age or gender, and not clearly associated with TCR counts. Elevated scores may reflect population-specific characteristics or weak biological factors.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Model diagnostic performance in early cancer detection</title>
<p>Building on the exceptional recognition performance of peripheral blood samples from early-stage breast cancer patients demonstrated in the previous study (AUC = 0.955), this research further validated the generalizability of DeepCaTCR for the early diagnosis of multiple cancer types. The model&#x2019;s capability to differentiate between tumor stages was systematically evaluated by collecting PBMC samples from patients with early (stage I&#x2013;II) and advanced (stage III&#x2013;IV) primary treatment. Additionally, independent healthy samples were collected as controls.</p>
<p>As illustrated in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>, the median cancer score for all early-stage cancers was significantly elevated compared to that of the healthy control group, as determined by the Wilcoxon test (p &lt; 0.05, applicable across all cancer types). Furthermore, Kendall&#x2019;s tau coefficient demonstrated a positive correlation between cancer scores and disease progression in both ovarian cancer (&#x3c4; = 0.629, p = 0.0047) and pancreatic cancer (&#x3c4; = 0.359, p = 0.094). DeepCaTCR achieved high AUCs for different stage cancers (stage I lung: 0.998; stage I RCC: 0.947; stage II pancreatic: 0.934, <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S5</bold>
</xref>), with specificity consistently &gt;86% across types.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Evaluation of DeepCaTCR&#x2019;s performance in detecting different stages of cancer. <bold>(A)</bold> The raincloud plot presents cancer scores for various cancer types at different stages in comparison to healthy controls (n = 58). Each cancer group is annotated with its type and stage, along with the sample size in parentheses. p-Values derived from Wilcoxon rank-sum tests, which compare each cancer group to healthy controls, are displayed above each comparison. Kendall&#x2019;s tau correlation coefficient was employed to evaluate the potential upward or downward trend in cancer score with increasing cancer stage. <bold>(B)</bold> ROC curves for DeepCaTCR across diverse cancer types and stages using the mean scorer. The legend specifies the cancer type and stage, along with the AUC value for each curve. <bold>(C)</bold> ROC curves for different models in early-stage lung cancer. The legend lists the model names and their associated AUC values. ROC, receiver operating characteristic; AUC, area under the receiver operating characteristic curve.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1625369-g004.tif">
<alt-text content-type="machine-generated">Panel A shows a violin plot comparing cancer scores across various stages of cancer and a healthy control group, with p-values indicating statistical significance. Panel B displays an ROC curve for cancer detection sensitivity and specificity, highlighting different stages of ovarian and pancreatic cancers. Panel C focuses on early-stage lung cancer, comparing diagnostic tools like DeepLION, BertTCR, DeepCAT, and DeepCaTCR, with their respective accuracy scores in parentheses.</alt-text>
</graphic>
</fig>
<p>In the context of early-stage lung cancer identification, DeepCaTCR demonstrated superior performance relative to all evaluated benchmarks, as detailed in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. DeepCaTCR achieved an AUC of 0.998 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4C</bold>
</xref>), surpassing DeepCAT&#x2019;s AUC of 0.912, while maintaining a balanced sensitivity of 100% and specificity of 98.3%. In contrast, DeepLION2 and BertTCR exhibited lower specificity at comparable sensitivity levels, with AUCs of 0.69 and 0.85, respectively. The MCC for DeepCaTCR was 0.945, compared to 0.719 for DeepCAT, highlighting its balanced classification performance.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Performance comparison with different models in early-stage lung cancer.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Model</th>
<th valign="middle" align="left">AUC</th>
<th valign="middle" align="left">ACC</th>
<th valign="middle" align="left">SEN</th>
<th valign="middle" align="left">SPE</th>
<th valign="middle" align="left">F1-score</th>
<th valign="middle" align="left">MCC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">DeepLION</td>
<td valign="middle" align="left">0.776</td>
<td valign="middle" align="left">0.809</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.776</td>
<td valign="middle" align="left">0.606</td>
<td valign="middle" align="left">0.581</td>
</tr>
<tr>
<td valign="middle" align="left">DeepLION2</td>
<td valign="middle" align="left">0.69</td>
<td valign="middle" align="left">0.735</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.69</td>
<td valign="middle" align="left">0.526</td>
<td valign="middle" align="left">0.496</td>
</tr>
<tr>
<td valign="middle" align="left">BertTCR</td>
<td valign="middle" align="left">0.85</td>
<td valign="middle" align="left">0.824</td>
<td valign="middle" align="left">0.9</td>
<td valign="middle" align="left">0.8103</td>
<td valign="middle" align="left">0.6</td>
<td valign="middle" align="left">0.5521</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCAT</td>
<td valign="middle" align="left">0.912</td>
<td valign="middle" align="left">0.897</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.879</td>
<td valign="middle" align="left">0.741</td>
<td valign="middle" align="left">0.719</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR</td>
<td valign="middle" align="left">0.998</td>
<td valign="middle" align="left">0.985</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.983</td>
<td valign="middle" align="left">0.952</td>
<td valign="middle" align="left">0.945</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>AUC, area under the receiver operating characteristic curve; ACC, accuracy; SEN, sensitivity; SPE, specificity; MCC, Matthews Correlation Coefficient.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Variance-based cancer predictor to enhance early cancer detection performance</title>
<p>Addressing the limitations inherent in the average scoring strategy for capturing the dynamic characteristics of the TCR repertoire, this study introduces a variance-based repertoire scoring method that markedly enhances the detection performance for early-stage cancers. While average scoring can indicate the overall tumor relevance of the TCR repertoire, it struggles to effectively characterize the heterogeneous features of TCR cancer score distribution during clonal amplification. Consequently, the variance scoring system developed in this study successfully captures the dynamic features of TCR clonal amplification during the early immune response by quantifying the degree of dispersion in cancer score distribution. As illustrated in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5A</bold>
</xref>, the distribution of cancer scores among early-stage patients with RCC, OV, PC, and lung cancer exhibited a more pronounced trend of intergroup segregation following the implementation of the variance scoring strategy (p &lt; 0.001). This development facilitated the identification of more discriminative features for the subsequent classification model.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Assessment of DeepCaTCR&#x2019;s efficacy in identifying early-stage cancer. <bold>(A)</bold> The raincloud plot illustrates cancer scores for early-stage cancer in contrast to healthy controls (n = 58). p-Values obtained from Wilcoxon rank-sum tests comparing each cancer group to healthy controls are indicated above each comparison. <bold>(B)</bold> Evaluation of DeepCaTCR&#x2019;s capability in early-stage cancer detection using the variance scorer, with ROC curves depicted for various cancer types. (<bold>C&#x2013;F</bold>) ROC curves for different models in early-stage cancer detection, with the legend providing model names and their corresponding AUC values. ROC, receiver operating characteristic; AUC, area under the receiver operating characteristic curve.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1625369-g005.tif">
<alt-text content-type="machine-generated">Panel A displays a box plot comparing cancer scores across various groups, including early-stage cancers and healthy controls. Panels B to F illustrate ROC curves for early-stage RCC, ovarian, pancreatic cancers, and all early-stage cancers, showing sensitivity versus specificity for different prediction models. Each panel includes a legend with model performance metrics, such as DeepCAT, BertTCR, and DeepCaTCR variants, indicated by their area under the curve (AUC) values.</alt-text>
</graphic>
</fig>
<p>The variance-based DeepCaTCR model demonstrates consistent enhancements in the AUC relative to baseline methodologies (<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>, <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5B&#x2013;F</bold>
</xref>). In the context of identifying pancreatic cancer patients, the AUC increased from 0.935 (mean scorer) to 0.972, representing an improvement of &#x394;AUC = +0.037, while specificity rose from 0.862 to 0.966. For ovarian cancer patient identification, the AUC increased from 0.852 to 0.931, maintaining high specificity (0.776 compared to 0.638 for the average classifier). In the multi-cancer identification task, the unified model achieved an AUC of 0.967, effectively balancing sensitivity (0.969) and specificity (0.897), thereby underscoring its applicability across various cancer types. These findings suggest that variance-based scoring can reduce the false-positive rate, as evidenced by the increased specificity for RCC (0.914 compared to 0.879 with the average scoring model) while preserving sensitivity, which is crucial for early detection.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Performance comparison with different models in early-stage cancer detection across multiple cancer types.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Disease</th>
<th valign="middle" align="left">Model</th>
<th valign="middle" align="left">AUC</th>
<th valign="middle" align="left">ACC</th>
<th valign="middle" align="left">SEN</th>
<th valign="middle" align="left">SPE</th>
<th valign="middle" align="left">F1-score</th>
<th valign="middle" align="left">MCC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="4" align="left">RCC</td>
<td valign="middle" align="left">DeepCAT</td>
<td valign="middle" align="left">0.829</td>
<td valign="middle" align="left">0.691</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.638</td>
<td valign="middle" align="left">0.488</td>
<td valign="middle" align="left">0.454</td>
</tr>
<tr>
<td valign="middle" align="left">BertTCR</td>
<td valign="middle" align="left">0.61</td>
<td valign="middle" align="left">0.794</td>
<td valign="middle" align="left">0.5</td>
<td valign="middle" align="left">0.845</td>
<td valign="middle" align="left">0.417</td>
<td valign="middle" align="left">0.302</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR<sup>Mean</sup>
</td>
<td valign="middle" align="left">0.947</td>
<td valign="middle" align="left">0.882</td>
<td valign="middle" align="left">0.9</td>
<td valign="middle" align="left">0.879</td>
<td valign="middle" align="left">0.692</td>
<td valign="middle" align="left">0.651</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR<sup>Variance</sup>
</td>
<td valign="middle" align="left">0.981</td>
<td valign="middle" align="left">0.927</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.914</td>
<td valign="middle" align="left">0.8</td>
<td valign="middle" align="left">0.781</td>
</tr>
<tr>
<td valign="middle" rowspan="4" align="left">OV</td>
<td valign="middle" align="left">DeepCAT</td>
<td valign="middle" align="left">0.813</td>
<td valign="middle" align="left">0.677</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.638</td>
<td valign="middle" align="left">0.4</td>
<td valign="middle" align="left">0.399</td>
</tr>
<tr>
<td valign="middle" align="left">BertTCR</td>
<td valign="middle" align="left">0.692</td>
<td valign="middle" align="left">0.677</td>
<td valign="middle" align="left">0.714</td>
<td valign="middle" align="left">0.672</td>
<td valign="middle" align="left">0.323</td>
<td valign="middle" align="left">0.248</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR<sup>Mean</sup>
</td>
<td valign="middle" align="left">0.852</td>
<td valign="middle" align="left">0.677</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.638</td>
<td valign="middle" align="left">0.4</td>
<td valign="middle" align="left">0.4</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR<sup>Variance</sup>
</td>
<td valign="middle" align="left">0.931</td>
<td valign="middle" align="left">0.8</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.776</td>
<td valign="middle" align="left">0.519</td>
<td valign="middle" align="left">0.521</td>
</tr>
<tr>
<td valign="middle" rowspan="4" align="left">PC</td>
<td valign="middle" align="left">DeepCAT</td>
<td valign="middle" align="left">0.759</td>
<td valign="middle" align="left">0.651</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.621</td>
<td valign="middle" align="left">0.313</td>
<td valign="middle" align="left">0.339</td>
</tr>
<tr>
<td valign="middle" align="left">BertTCR</td>
<td valign="middle" align="left">0.628</td>
<td valign="middle" align="left">0.603</td>
<td valign="middle" align="left">0.8</td>
<td valign="middle" align="left">0.586</td>
<td valign="middle" align="left">0.242</td>
<td valign="middle" align="left">0.21</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR<sup>Mean</sup>
</td>
<td valign="middle" align="left">0.935</td>
<td valign="middle" align="left">0.873</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.862</td>
<td valign="middle" align="left">0.556</td>
<td valign="middle" align="left">0.576</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR<sup>Variance</sup>
</td>
<td valign="middle" align="left">0.972</td>
<td valign="middle" align="left">0.968</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.966</td>
<td valign="middle" align="left">0.833</td>
<td valign="middle" align="left">0.831</td>
</tr>
<tr>
<td valign="middle" rowspan="4" align="left">All cancers</td>
<td valign="middle" align="left">DeepCAT</td>
<td valign="middle" align="left">0.841</td>
<td valign="middle" align="left">0.756</td>
<td valign="middle" align="left">1.0</td>
<td valign="middle" align="left">0.621</td>
<td valign="middle" align="left">0.744</td>
<td valign="middle" align="left">0.607</td>
</tr>
<tr>
<td valign="middle" align="left">BertTCR</td>
<td valign="middle" align="left">0.706</td>
<td valign="middle" align="left">0.7</td>
<td valign="middle" align="left">0.75</td>
<td valign="middle" align="left">0.672</td>
<td valign="middle" align="left">0.64</td>
<td valign="middle" align="left">0.405</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR<sup>Mean</sup>
</td>
<td valign="middle" align="left">0.94</td>
<td valign="middle" align="left">0.867</td>
<td valign="middle" align="left">0.875</td>
<td valign="middle" align="left">0.862</td>
<td valign="middle" align="left">0.824</td>
<td valign="middle" align="left">0.72</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR<sup>Variance</sup>
</td>
<td valign="middle" align="left">0.967</td>
<td valign="middle" align="left">0.922</td>
<td valign="middle" align="left">0.969</td>
<td valign="middle" align="left">0.897</td>
<td valign="middle" align="left">0.899</td>
<td valign="middle" align="left">0.842</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>AUC, area under the receiver operating characteristic curve; ACC, accuracy; SEN, sensitivity; SPE, specificity; MCC, Matthews Correlation Coefficient; RCC, renal cell carcinoma; OV, ovarian cancer; PC, pancreatic cancer.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>At clinically actionable specificity thresholds, variance scores demonstrated exceptional performance (<xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>). Specifically, at a specificity level exceeding 98%, variance scores achieved a sensitivity of 62.5%, compared to 53.1% for mean scores, thereby significantly surpassing the performance of DeepCAT (9.4%) and BertTCR (0%). When the specificity threshold was set above 95%, the sensitivity of variance scores increased to 81.3%, in contrast to 75% for the mean score method, indicating their reliability in low-prevalence screening scenarios. Furthermore, the variance-based scoring method exhibited greater robustness in terms of AUC stability, as evidenced by a narrower 95% confidence interval (0.934&#x2013;0.999) compared to that of the mean-based scoring method (0.895&#x2013;0.986). The efficacy of the variance-based scoring method may be attributed to its capacity to quantify TCR clonal diversity during the early stages of tumor development, a characteristic that is not captured by mean-based methods.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Model performance comparison for early-stage cancer detection across different specificity thresholds.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Model</th>
<th valign="middle" align="left">AUC (95% CI)</th>
<th valign="middle" align="left">Sensitivity (specificity &gt; 98%)</th>
<th valign="middle" align="left">Sensitivity (specificity &gt; 95%)</th>
<th valign="middle" align="left">Sensitivity (specificity &gt; 90%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">DeepCAT</td>
<td valign="middle" align="left">0.841 (0.760&#x2013;0.921)</td>
<td valign="middle" align="left">0.094</td>
<td valign="middle" align="left">0.156</td>
<td valign="middle" align="left">0.312</td>
</tr>
<tr>
<td valign="middle" align="left">BertTCR</td>
<td valign="middle" align="left">0.706 (0.587&#x2013;0.825)</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="left">0</td>
<td valign="middle" align="left">0.25</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR<sup>Mean</sup>
</td>
<td valign="middle" align="left">0.94 (0.895&#x2013;0.986)</td>
<td valign="middle" align="left">0.531</td>
<td valign="middle" align="left">0.75</td>
<td valign="middle" align="left">0.781</td>
</tr>
<tr>
<td valign="middle" align="left">DeepCaTCR<sup>Variance</sup>
</td>
<td valign="middle" align="left">0.967 (0.934&#x2013;0.999)</td>
<td valign="middle" align="left">0.625</td>
<td valign="middle" align="left">0.813</td>
<td valign="middle" align="left">0.875</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>AUC, area under the receiver operating characteristic curve.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_6">
<label>3.6</label>
<title>Biological insights into TCR sequences predicted by DeepCaTCR</title>
<p>To elucidate the biological relevance of TCR sequences predicted by DeepCaTCR, key motifs, their functional significance, and overlap with known cancer-associated TCRs were analyzed (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>). DeepCaTCR identified key amino acid motifs in TCR sequences from the validation set and assigned importance scores to each motif (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>). Visualization results show that larger and darker residues correspond to higher importance scores, indicating that these sequence patterns may play a key role in antigen recognition. Among TCRs with high prediction confidence (score &gt; 0.95), certain motifs were highly recurrent (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>). The most frequent motifs included &#x201c;CSAR&#x201d; (140 occurrences), &#x201c;CASP&#x201d; (45 occurrences), and &#x201c;PG&#x201d; (35 occurrences). These motifs may represent conserved structural or functional elements in cancer-associated TCRs. A heatmap analysis revealed that DeepCaTCR-identified motifs are enriched in the McPAS-TCR (<xref ref-type="bibr" rid="B45">45</xref>) database (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6C</bold>
</xref>). &#x201c;ASS&#x201d; (24,730 occurrences) and &#x201c;ASSL&#x201d; (5,759 occurrences) were among the most frequent motifs in McPAS-TCR, aligning with their high frequency in DeepCaTCR predictions. Other motifs like &#x201c;AG&#x201d; (4,831 occurrences) and &#x201c;EA&#x201d; (3,997 occurrences) further validated the biological relevance of DeepCaTCR&#x2019;s predictions.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Visualization of motifs and biological insights in DeepCaTCR predicted TCR sequences. <bold>(A)</bold> The visualization of key motifs and their corresponding importance scores within TCR sequences derived from the test set. For each of the 10 representative TCR sequences, the amino acid residues identified as critical motifs by DeepCaTCR are depicted. The size and color intensity of each residue are indicative of its importance score, with larger and darker residues signifying higher scores. <bold>(B)</bold> This panel presents the frequency of key motifs in TCRs with high prediction confidence (prediction score &gt; 0.95). The bar plot provides a summary of the occurrence of top-scoring motifs across TCR sequences with high prediction confidence. <bold>(C)</bold> Frequency of DeepCaTCR-identified key motifs in McPAS-TCR database. A heatmap shows the prevalence of predicted motifs in McPAS-TCR, with darker colors indicating higher occurrence frequencies. <bold>(D)</bold> The overlap between the top 22 high-scoring TCR sequences and known cancer-associated motifs sourced from public databases (TCRdb, VDJdb, and McPAS-TCR). Blue bars represent the number of TCRs that match known cancer-associated motifs, while white bars indicate novel TCRs that do not have matches in the databases. TCR, T-cell receptor.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1625369-g006.tif">
<alt-text content-type="machine-generated">Diagram with four panels showing peptide sequences and their frequencies. Panel A shows a list of sequences with varying colors. Panel B is a bar chart of sequence frequencies, highlighting &#x201c;CSAR&#x201d; as most frequent. Panel C presents a scatter plot of sequences versus frequency on a logarithmic scale. Panel D is a heatmap of sequence frequencies across three databases: McPAS-TCR, VDJdb, and TCRdb, with varying shades representing different frequencies.</alt-text>
</graphic>
</fig>
<p>We initially conducted a search for the top 22 high-scoring TCRs (score &gt; 0.98) across three major databases [TCRdb (<xref ref-type="bibr" rid="B46">46</xref>), VDJdb (<xref ref-type="bibr" rid="B47">47</xref>), and McPAS-TCR (<xref ref-type="bibr" rid="B45">45</xref>)] but did not identify any exact matches among cancer-associated TCRs. This outcome is likely attributable to the exceptionally high diversity of TCR sequences. Considering the significant heterogeneity across cancer types, the absence of these 22 TCRs in existing databases is biologically plausible. To evaluate potential partial matches, we applied various mismatch tolerance criteria tailored to each database&#x2019;s functionalities. In TCRdb, we recorded near-matches with up to two amino acid mismatches, as provided by the database. VDJdb allowed for the extraction of similar sequences with an Informativeness score of 8 or higher, indicating high-confidence hits. For McPAS-TCR, we conducted local searches using Python scripts, identifying sequences with up to four mismatches, although no hits were found with two or fewer mismatches.</p>
<p>A detailed breakdown is provided in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6D</bold>
</xref>, where blue blocks denote partial database matches, such as &#x201c;CSVEDRRRTGYTEAFF&#x201d; with five matches in TCRdb and one in McPAS-TCR. The analysis revealed that 11 TCRs (50%) exhibited partial matches in at least one database. For instance, the TCR sequence &#x201c;CASSSGLAVPCNEQFF&#x201d; demonstrated four matches in TCRdb and two high-confidence matches in VDJdb, in addition to seven hits in McPAS-TCR. Another TCR, &#x201c;CSAHPGGLAGAEQYF&#x201d;, was found to have two matches in TCRdb and seven in McPAS-TCR. Conversely, 11 TCRs (50%) did not exhibit matches in any of the databases under the specified criteria, exemplified by sequences such as &#x201c;CSAPRDSLRRADEQYF&#x201d; and &#x201c;CSARPRGPLAAEAFF&#x201d;, suggesting that these may represent previously uncharacterized cancer-reactive TCRs.</p>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<label>4</label>
<title>Conclusion</title>
<p>In this study, the DeepCaTCR deep learning framework was developed to enhance the recognition specificity of cancer-associated TCRs. This was achieved by integrating a one-dimensional variable convolutional kernel, bidirectional long- and short-term memory units, and a self-attention mechanism, resulting in a discriminative efficacy with an AUC of 0.863 in cross-cancer validation. Additionally, the proposed variance scoring strategy, which is based on TCR&#x3b2; CDR3 clonal amplification, improved the sensitivity of early-stage cancer detection in peripheral blood to 62.5% by quantifying the heterogeneous features of the immunohistochemical repertoire. This approach achieved an AUC of 0.967 in pan-cancer screening, offering a novel solution to the technical challenge of detecting weak tumor signals in liquid biopsy.</p>
</sec>
<sec id="s5" sec-type="discussion">
<label>5</label>
<title>Discussion</title>
<p>In this study, we developed DeepCaTCR, a deep learning-based framework for TCR repertoire analysis, aimed at improving the efficacy of early cancer detection. A key innovation of this framework is the introduction of a variance-based repertoire scoring strategy, which addresses the limitations of traditional average scoring methods in capturing the dynamic characteristics of immune responses. This novel approach not only enhances the characterization of these dynamics but also establishes a new technical paradigm for pan-cancer early screening. The superior performance of the variance scoring strategy is attributed to its precise modeling of TCR clonal amplification biology. During the initial stages of tumorigenesis, nascent antigen-specific T cells undergo clonal expansion, leading to a highly heterogeneous TCR distribution profile. Our findings indicate that this dynamic evolutionary process is reflected in a significantly greater dispersion in cancer score distribution. In contrast, the conventional mean-value method, by smoothing the data, diminishes the detection sensitivity of this critical biological signal. Through rigorous mathematical modeling and clinical validation, we established a quantitative association between the variance of the TCR distribution and the strength of the tumor immune response.</p>
<p>Despite the advancements achieved, several limitations persist in this study. First, the existing validation predominantly addresses solid tumors, and its applicability to hematological malignancies remains unverified. Second, the occurrence of false positives observed in the HCMV-infected cohort underscores the necessity for an improved background filtering system tailored to infected backgrounds. Lastly, this study utilized retrospective data, necessitating prospective cohort studies to substantiate clinical efficacy. Future research will concentrate on 1) integrating epitope prediction data to refine the variance scoring algorithm, 2) developing a dynamic scoring model informed by longitudinal surveillance, and 3) creating a clinical decision support system to accompany these advancements.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>Ethical approval was not required for the study involving humans in accordance with the local legislation and institutional requirements. Written informed consent to participate in this study was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>YT: Conceptualization, Data curation, Formal analysis, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. XL: Data curation, Formal analysis, Writing &#x2013; review &amp; editing. BL: Conceptualization, Formal analysis, Funding acquisition, Supervision, Writing &#x2013; review &amp; editing. DP: Conceptualization, Writing &#x2013; review &amp; editing. QL: Visualization, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. This study was funded by National Natural Science Foundation of China (Grant Nos. 62362027 and 62362028), National Key R&amp;D Program of China (No. 2020YFB2104400), Natural Science Foundation of Hainan, China (Grant Nos. 824MS063, 824MS062, and 122MS055), and Program of Graduate Education and Teaching Reform in Hainan, China (Grant No. Hnjg2023-54).</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We thank Dr. Yideng Cai and Professor Qinghua Jiang of Harbin Institute of Technology for their help with this work.</p>
</ack>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Authors XL was employed by the company China Unicom Hainan Industrial Internet Co. Ltd.</p>
<p>The remaining author declares that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s13" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fonc.2025.1625369/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fonc.2025.1625369/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Crosby</surname> <given-names>D</given-names>
</name>
<name>
<surname>Bhatia</surname> <given-names>S</given-names>
</name>
<name>
<surname>Brindle</surname> <given-names>KM</given-names>
</name>
<name>
<surname>Coussens</surname> <given-names>LM</given-names>
</name>
<name>
<surname>Dive</surname> <given-names>C</given-names>
</name>
<name>
<surname>Emberton</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Early detection of cancer</article-title>. <source>Science</source>. (<year>2022</year>) <volume>375</volume>:<elocation-id>eaay9040</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.aay9040</pub-id>, PMID: <pub-id pub-id-type="pmid">35298272</pub-id></citation></ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>X</given-names>
</name>
<name>
<surname>Gole</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gore</surname> <given-names>A</given-names>
</name>
<name>
<surname>He</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Min</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Non-invasive early detection of cancer four years before conventional diagnosis using a blood test</article-title>. <source>Nat Commun</source>. (<year>2020</year>) <volume>11</volume>:<fpage>3475</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-020-17316-z</pub-id>, PMID: <pub-id pub-id-type="pmid">32694610</pub-id></citation></ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Henschke</surname> <given-names>C</given-names>
</name>
<name>
<surname>Huber</surname> <given-names>R</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>L</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>D</given-names>
</name>
<name>
<surname>Cavic</surname> <given-names>M</given-names>
</name>
<name>
<surname>Schmidt</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Perspective on management of low-dose computed tomography findings on low-dose computed tomography examinations for lung cancer screening. From the international association for the study of lung cancer early detection and screening committee</article-title>. <source>J Thorac Oncol</source>. (<year>2024</year>) <volume>19</volume>:<page-range>565&#x2013;80</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jtho.2023.11.013</pub-id>, PMID: <pub-id pub-id-type="pmid">37979778</pub-id></citation></ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maier</surname> <given-names>A</given-names>
</name>
<name>
<surname>Teunissen</surname> <given-names>AJP</given-names>
</name>
<name>
<surname>Nauta</surname> <given-names>SA</given-names>
</name>
<name>
<surname>Lutgens</surname> <given-names>E</given-names>
</name>
<name>
<surname>Fayad</surname> <given-names>ZA</given-names>
</name>
<name>
<surname>van Leent</surname> <given-names>MMT</given-names>
</name>
</person-group>. <article-title>Uncovering atherosclerotic cardiovascular disease by PET imaging</article-title>. <source>Nat Rev Cardiol</source>. (<year>2024</year>) <volume>21</volume>:<page-range>632&#x2013;51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41569-024-01009-x</pub-id>, PMID: <pub-id pub-id-type="pmid">38575752</pub-id></citation></ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karlas</surname> <given-names>A</given-names>
</name>
<name>
<surname>Pleitez</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Aguirre</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ntziachristos</surname> <given-names>V</given-names>
</name>
</person-group>. <article-title>Optoacoustic imaging in endocrinology and metabolism</article-title>. <source>Nat Rev Endocrinol</source>. (<year>2021</year>) <volume>17</volume>:<page-range>323&#x2013;35</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41574-021-00482-5</pub-id>, PMID: <pub-id pub-id-type="pmid">33875856</pub-id></citation></ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ando</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Dbouk</surname> <given-names>M</given-names>
</name>
<name>
<surname>Yoshida</surname> <given-names>T</given-names>
</name>
<name>
<surname>Saba</surname> <given-names>H</given-names>
</name>
<name>
<surname>Diwan</surname> <given-names>EA</given-names>
</name>
<name>
<surname>Yoshida</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Using tumor marker gene variants to improve the diagnostic accuracy of DUPAN-2 and carbohydrate antigen 19&#x2013;9 for pancreatic cancer</article-title>. <source>J Clin Oncol</source>. (<year>2024</year>) <volume>42</volume>:<page-range>2196&#x2013;206</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/JCO.23.01573</pub-id>, PMID: <pub-id pub-id-type="pmid">38457748</pub-id></citation></ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Preethi</surname> <given-names>KA</given-names>
</name>
<name>
<surname>Selvakumar</surname> <given-names>SC</given-names>
</name>
<name>
<surname>Ross</surname> <given-names>K</given-names>
</name>
<name>
<surname>Jayaraman</surname> <given-names>S</given-names>
</name>
<name>
<surname>Tusubira</surname> <given-names>D</given-names>
</name>
<name>
<surname>Sekar</surname> <given-names>D</given-names>
</name>
</person-group>. <article-title>Liquid biopsy: Exosomal microRNAs as novel diagnostic and prognostic biomarkers in cancer</article-title>. <source>Mol Cancer</source>. (<year>2022</year>) <volume>21</volume>:<fpage>54</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12943-022-01525-9</pub-id>, PMID: <pub-id pub-id-type="pmid">35172817</pub-id></citation></ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>K</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Xin</surname> <given-names>J</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Circular RNAs in body fluids as cancer biomarkers: the new frontier of liquid biopsies</article-title>. <source>Mol Cancer</source>. (<year>2021</year>) <volume>20</volume>:<fpage>13</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12943-020-01298-z</pub-id>, PMID: <pub-id pub-id-type="pmid">33430880</pub-id></citation></ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>MC</given-names>
</name>
<name>
<surname>Oxnard</surname> <given-names>GR</given-names>
</name>
<name>
<surname>Klein</surname> <given-names>EA</given-names>
</name>
<name>
<surname>Swanton</surname> <given-names>C</given-names>
</name>
<name>
<surname>Seiden</surname> <given-names>MV</given-names>
</name>
<collab>Consortium C</collab>
</person-group>. <article-title>Sensitive and specific multi-cancer detection and localization using methylation signatures in cell-free DNA</article-title>. <source>Ann Oncol</source>. (<year>2020</year>) <volume>31</volume>:<page-range>745&#x2013;59</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.annonc.2020.02.011</pub-id>, PMID: <pub-id pub-id-type="pmid">33506766</pub-id></citation></ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dyikanov</surname> <given-names>D</given-names>
</name>
<name>
<surname>Zaitsev</surname> <given-names>A</given-names>
</name>
<name>
<surname>Vasileva</surname> <given-names>T</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>I</given-names>
</name>
<name>
<surname>Sokolov</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Bolshakov</surname> <given-names>ES</given-names>
</name>
<etal/>
</person-group>. <article-title>Comprehensive peripheral blood immunoprofiling reveals five immunotypes with immunotherapy response characteristics in patients with cancer</article-title>. <source>Cancer Cell</source>. (<year>2024</year>) <volume>42</volume>:<page-range>759&#x2013;79</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ccell.2024.04.008</pub-id>, PMID: <pub-id pub-id-type="pmid">38744245</pub-id></citation></ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spasic</surname> <given-names>M</given-names>
</name>
<name>
<surname>Ogayo</surname> <given-names>ER</given-names>
</name>
<name>
<surname>Parsons</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Mittendorf</surname> <given-names>EA</given-names>
</name>
<name>
<surname>Galen</surname> <given-names>P</given-names>
</name>
<name>
<surname>McAllister</surname> <given-names>SS</given-names>
</name>
</person-group>. <article-title>Spectral flow cytometry methods and pipelines for comprehensive immunoprofiling of human peripheral blood and bone marrow</article-title>. <source>Cancer Res Commun</source>. (<year>2024</year>) <volume>4</volume>:<fpage>895</fpage>&#x2013;<lpage>910</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/2767-9764.CRC-23-0357</pub-id>, PMID: <pub-id pub-id-type="pmid">38466569</pub-id></citation></ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Enfield</surname> <given-names>KSS</given-names>
</name>
<name>
<surname>Colliver</surname> <given-names>E</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>C</given-names>
</name>
<name>
<surname>Magness</surname> <given-names>A</given-names>
</name>
<name>
<surname>Moore</surname> <given-names>DA</given-names>
</name>
<name>
<surname>Sivakumar</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Spatial architecture of myeloid and T cells orchestrates immune evasion and clinical outcome in lung cancer</article-title>. <source>Cancer Discov</source>. (<year>2024</year>) <volume>14</volume>:<page-range>1018&#x2013;47</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/2159-8290.CD-23-1380</pub-id>, PMID: <pub-id pub-id-type="pmid">38581685</pub-id></citation></ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>H</given-names>
</name>
<name>
<surname>Zandberg</surname> <given-names>DP</given-names>
</name>
<name>
<surname>Kulkarni</surname> <given-names>A</given-names>
</name>
<name>
<surname>Chiosea</surname> <given-names>SI</given-names>
</name>
<name>
<surname>Santos</surname> <given-names>PM</given-names>
</name>
<name>
<surname>Isett</surname> <given-names>BR</given-names>
</name>
<etal/>
</person-group>. <article-title>Distinct CD8(+) T cell dynamics associate with response to neoadjuvant cancer immunotherapies</article-title>. <source>Cancer Cell</source>. (<year>2025</year>) <volume>43</volume>:<page-range>757&#x2013;75</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ccell.2025.02.026</pub-id>, PMID: <pub-id pub-id-type="pmid">40086437</pub-id></citation></ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pai</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Hellmann</surname> <given-names>MD</given-names>
</name>
<name>
<surname>Sauter</surname> <given-names>J</given-names>
</name>
<name>
<surname>Mattar</surname> <given-names>M</given-names>
</name>
<name>
<surname>Rizvi</surname> <given-names>H</given-names>
</name>
<name>
<surname>Woo</surname> <given-names>HJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Lineage tracing reveals clonal progenitors and long-term persistence of tumor-specific T cells during immune checkpoint blockade</article-title>. <source>Cancer Cell</source>. (<year>2023</year>) <volume>41</volume>:<page-range>776&#x2013;90</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ccell.2023.03.009</pub-id>, PMID: <pub-id pub-id-type="pmid">37001526</pub-id></citation></ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lagattuta</surname> <given-names>KA</given-names>
</name>
<name>
<surname>Kang</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Nathan</surname> <given-names>A</given-names>
</name>
<name>
<surname>Pauken</surname> <given-names>KE</given-names>
</name>
<name>
<surname>Jonsson</surname> <given-names>AH</given-names>
</name>
<name>
<surname>Rao</surname> <given-names>DA</given-names>
</name>
<etal/>
</person-group>. <article-title>Repertoire analyses reveal T cell antigen receptor sequence features that influence T cell fate</article-title>. <source>Nat Immunol</source>. (<year>2022</year>) <volume>23</volume>:<page-range>446&#x2013;57</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41590-022-01129-x</pub-id>, PMID: <pub-id pub-id-type="pmid">35177831</pub-id></citation></ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Genolet</surname> <given-names>R</given-names>
</name>
<name>
<surname>Bobisse</surname> <given-names>S</given-names>
</name>
<name>
<surname>Chiffelle</surname> <given-names>J</given-names>
</name>
<name>
<surname>Arnaud</surname> <given-names>M</given-names>
</name>
<name>
<surname>Petremand</surname> <given-names>R</given-names>
</name>
<name>
<surname>Queiroz</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>TCR sequencing and cloning methods for repertoire analysis and isolation of tumor-reactive TCRs</article-title>. <source>Cell Rep Methods</source>. (<year>2023</year>) <volume>3</volume>:<fpage>100459</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.crmeth.2023.100459</pub-id>, PMID: <pub-id pub-id-type="pmid">37159666</pub-id></citation></ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wong</surname> <given-names>C</given-names>
</name>
<name>
<surname>Li</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>AutoCAT: automated cancer-associated TCRs discovery from TCR-seq data</article-title>. <source>Bioinformatics</source>. (<year>2022</year>) <volume>38</volume>:<page-range>589&#x2013;91</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btab661</pub-id>, PMID: <pub-id pub-id-type="pmid">34529039</pub-id></citation></ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kidman</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zemek</surname> <given-names>RM</given-names>
</name>
<name>
<surname>Sidhom</surname> <given-names>JW</given-names>
</name>
<name>
<surname>Correa</surname> <given-names>D</given-names>
</name>
<name>
<surname>Principe</surname> <given-names>N</given-names>
</name>
<name>
<surname>Sheikh</surname> <given-names>F</given-names>
</name>
<etal/>
</person-group>. <article-title>Immune checkpoint therapy responders display early clonal expansion of tumor infiltrating lymphocytes</article-title>. <source>Oncoimmunology</source>. (<year>2024</year>) <volume>13</volume>:<fpage>2345859</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/2162402X.2024.2345859</pub-id>, PMID: <pub-id pub-id-type="pmid">38686178</pub-id></citation></ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beshnova</surname> <given-names>D</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>J</given-names>
</name>
<name>
<surname>Onabolu</surname> <given-names>O</given-names>
</name>
<name>
<surname>Moon</surname> <given-names>B</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>W</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>
<italic>De novo</italic> prediction of cancer-associated T cell receptors for noninvasive cancer detection</article-title>. <source>Sci Transl Med</source>. (<year>2020</year>) <volume>12</volume>:<fpage>aaz3738</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/scitranslmed.aaz3738</pub-id>, PMID: <pub-id pub-id-type="pmid">32817363</pub-id></citation></ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Qian</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>X</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>DeepLION: deep multi-instance learning improves the prediction of cancer-associated T cell receptors for accurate cancer detection</article-title>. <source>Front Genet</source>. (<year>2022</year>) <volume>13</volume>:<elocation-id>860510</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fgene.2022.860510</pub-id>, PMID: <pub-id pub-id-type="pmid">35601486</pub-id></citation></ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qian</surname> <given-names>X</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>G</given-names>
</name>
<name>
<surname>Li</surname> <given-names>F</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>X</given-names>
</name>
<etal/>
</person-group>. <article-title>DeepLION2: deep multi-instance contrastive learning framework enhancing the prediction of cancer-associated T cell receptors by attention strategy on motifs</article-title>. <source>Front Immunol</source>. (<year>2024</year>) <volume>15</volume>:<elocation-id>1345586</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2024.1345586</pub-id>, PMID: <pub-id pub-id-type="pmid">38515756</pub-id></citation></ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>M</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>S</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>N</given-names>
</name>
<etal/>
</person-group>. <article-title>BertTCR: a Bert-based deep learning framework for predicting cancer-related immune status based on T cell receptor repertoire</article-title>. <source>Brief Bioinform</source>. (<year>2024</year>) <volume>25</volume>:<fpage>bbae420</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bib/bbae420</pub-id>, PMID: <pub-id pub-id-type="pmid">39177262</pub-id></citation></ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>M</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>C</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>P</given-names>
</name>
<name>
<surname>Xue</surname> <given-names>G</given-names>
</name>
<etal/>
</person-group>. <article-title>The deep learning framework iCanTCR enables early cancer detection using the T-cell receptor repertoire in peripheral blood</article-title>. <source>Cancer Res</source>. (<year>2024</year>) <volume>84</volume>:<page-range>1915&#x2013;28</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/0008-5472.CAN-23-0860</pub-id>, PMID: <pub-id pub-id-type="pmid">38536129</pub-id></citation></ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>B</given-names>
</name>
<name>
<surname>Li</surname> <given-names>T</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>B</given-names>
</name>
<name>
<surname>Dou</surname> <given-names>R</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>JS</given-names>
</name>
<etal/>
</person-group>. <article-title>Ultrasensitive detection of TCR hypervariable-region sequences in solid-tissue RNA-seq data</article-title>. <source>Nat Genet</source>. (<year>2017</year>) <volume>49</volume>:<page-range>482&#x2013;3</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ng.3820</pub-id>, PMID: <pub-id pub-id-type="pmid">28358132</pub-id></citation></ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tomczak</surname> <given-names>K</given-names>
</name>
<name>
<surname>Czerwinska</surname> <given-names>P</given-names>
</name>
<name>
<surname>Wiznerowicz</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>The Cancer Genome Atlas (TCGA): an immeasurable source of knowledge</article-title>. <source>Contemp Oncol (Pozn)</source>. (<year>2015</year>) <volume>19</volume>:<page-range>A68&#x2013;77</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.5114/wo.2014.47136</pub-id>, PMID: <pub-id pub-id-type="pmid">25691825</pub-id></citation></ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Emerson</surname> <given-names>RO</given-names>
</name>
<name>
<surname>DeWitt</surname> <given-names>WS</given-names>
</name>
<name>
<surname>Vignali</surname> <given-names>M</given-names>
</name>
<name>
<surname>Gravley</surname> <given-names>J</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>JK</given-names>
</name>
<name>
<surname>Osborne</surname> <given-names>EJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Immunosequencing identifies signatures of cytomegalovirus exposure history and HLA-mediated effects on the T cell repertoire</article-title>. <source>Nat Genet</source>. (<year>2017</year>) <volume>49</volume>:<page-range>659&#x2013;65</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ng.3822</pub-id>, PMID: <pub-id pub-id-type="pmid">28369038</pub-id></citation></ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>J</given-names>
</name>
<name>
<surname>Shukla</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Investigation of antigen-specific T-cell receptor clusters in human cancers</article-title>. <source>Clin Cancer Res</source>. (<year>2020</year>) <volume>26</volume>:<page-range>1359&#x2013;71</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/1078-0432.CCR-19-3249</pub-id>, PMID: <pub-id pub-id-type="pmid">31831563</pub-id></citation></ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beausang</surname> <given-names>JF</given-names>
</name>
<name>
<surname>Wheeler</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Chan</surname> <given-names>NH</given-names>
</name>
<name>
<surname>Hanft</surname> <given-names>VR</given-names>
</name>
<name>
<surname>Dirbas</surname> <given-names>FM</given-names>
</name>
<name>
<surname>Jeffrey</surname> <given-names>SS</given-names>
</name>
<etal/>
</person-group>. <article-title>T cell receptor sequencing of early-stage breast cancer tumors identify altered clonal structure of the T cell repertoire</article-title>. <source>Proc Natl Acad Sci U S A</source>. (<year>2017</year>) <volume>114</volume>:<page-range>E10409&#x2013;17</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1713863114</pub-id>, PMID: <pub-id pub-id-type="pmid">29138313</pub-id></citation></ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mansfield</surname> <given-names>AS</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>H</given-names>
</name>
<name>
<surname>Sutor</surname> <given-names>S</given-names>
</name>
<name>
<surname>Sarangi</surname> <given-names>V</given-names>
</name>
<name>
<surname>Nair</surname> <given-names>A</given-names>
</name>
<name>
<surname>Davila</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Contraction of T cell richness in lung cancer brain metastases</article-title>. <source>Sci Rep</source>. (<year>2018</year>) <volume>8</volume>:<fpage>2171</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-018-20622-8</pub-id>, PMID: <pub-id pub-id-type="pmid">29391594</pub-id></citation></ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tumeh</surname> <given-names>PC</given-names>
</name>
<name>
<surname>Harview</surname> <given-names>CL</given-names>
</name>
<name>
<surname>Yearley</surname> <given-names>JH</given-names>
</name>
<name>
<surname>Shintaku</surname> <given-names>IP</given-names>
</name>
<name>
<surname>Taylor</surname> <given-names>EJM</given-names>
</name>
<name>
<surname>Robert</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>PD-1 blockade induces responses by inhibiting adaptive immune resistance</article-title>. <source>Nature</source>. (<year>2014</year>) <volume>515</volume>:<page-range>568&#x2013;71</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature13954</pub-id>, PMID: <pub-id pub-id-type="pmid">25428505</pub-id></citation></ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stromnes</surname> <given-names>IM</given-names>
</name>
<name>
<surname>Hulbert</surname> <given-names>A</given-names>
</name>
<name>
<surname>Pierce</surname> <given-names>RH</given-names>
</name>
<name>
<surname>Greenberg</surname> <given-names>PD</given-names>
</name>
<name>
<surname>Hingorani</surname> <given-names>SR</given-names>
</name>
</person-group>. <article-title>T-cell localization, activation, and clonal expansion in human pancreatic ductal adenocarcinoma</article-title>. <source>Cancer Immunol Res</source>. (<year>2017</year>) <volume>5</volume>:<page-range>978&#x2013;91</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/2326-6066.CIR-16-0322</pub-id>, PMID: <pub-id pub-id-type="pmid">29066497</pub-id></citation></ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robert</surname> <given-names>L</given-names>
</name>
<name>
<surname>Tsoi</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Emerson</surname> <given-names>R</given-names>
</name>
<name>
<surname>Homet</surname> <given-names>B</given-names>
</name>
<name>
<surname>Chodon</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>CTLA4 blockade broadens the peripheral T-cell receptor repertoire</article-title>. <source>Clin Cancer Res</source>. (<year>2014</year>) <volume>20</volume>:<page-range>2424&#x2013;32</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/1078-0432.CCR-13-2648</pub-id>, PMID: <pub-id pub-id-type="pmid">24583799</pub-id></citation></ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Emerson</surname> <given-names>RO</given-names>
</name>
<name>
<surname>Sherwood</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Rieder</surname> <given-names>MJ</given-names>
</name>
<name>
<surname>Guenthoer</surname> <given-names>J</given-names>
</name>
<name>
<surname>Williamson</surname> <given-names>DW</given-names>
</name>
<name>
<surname>Christopher</surname> <given-names>SC</given-names>
</name>
<etal/>
</person-group>. <article-title>High-throughput sequencing of T-cell receptors reveals a homogeneous repertoire of tumour-infiltrating lymphocytes in ovarian cancer</article-title>. <source>J Pathol</source>. (<year>2013</year>) <volume>231</volume>:<page-range>433&#x2013;40</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/path.4260</pub-id>, PMID: <pub-id pub-id-type="pmid">24027095</pub-id></citation></ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>DeWitt</surname> <given-names>WS</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>KKQ</given-names>
</name>
<name>
<surname>Wilburn</surname> <given-names>DB</given-names>
</name>
<name>
<surname>Sherwood</surname> <given-names>A</given-names>
</name>
<name>
<surname>Vignail</surname> <given-names>M</given-names>
</name>
<name>
<surname>Day</surname> <given-names>CL</given-names>
</name>
<etal/>
</person-group>. <article-title>A diverse lipid antigen-specific TCR repertoire is clonally expanded during active tuberculosis</article-title>. <source>J Immunol</source>. (<year>2018</year>) <volume>201</volume>:<page-range>888&#x2013;96</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.1800186</pub-id>, PMID: <pub-id pub-id-type="pmid">29914888</pub-id></citation></ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Snyder</surname> <given-names>A</given-names>
</name>
<name>
<surname>Nathanson</surname> <given-names>T</given-names>
</name>
<name>
<surname>Funt</surname> <given-names>SA</given-names>
</name>
<name>
<surname>Ahuja</surname> <given-names>A</given-names>
</name>
<name>
<surname>Novik</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Hellmann</surname> <given-names>MD</given-names>
</name>
<etal/>
</person-group>. <article-title>Contribution of systemic and somatic factors to clinical response and resistance to PD-L1 blockade in urothelial cancer: An exploratory multi-omic analysis</article-title>. <source>PloS Med</source>. (<year>2017</year>) <volume>14</volume>:<elocation-id>e1002309</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pmed.1002309</pub-id>, PMID: <pub-id pub-id-type="pmid">28552987</pub-id></citation></ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hsu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sedighim</surname> <given-names>S</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>T</given-names>
</name>
<name>
<surname>Antonios</surname> <given-names>JP</given-names>
</name>
<name>
<surname>Everson</surname> <given-names>RG</given-names>
</name>
<name>
<surname>Tucker</surname> <given-names>AM</given-names>
</name>
<etal/>
</person-group>. <article-title>TCR sequencing can identify and track glioma-infiltrating T cells after DC vaccination</article-title>. <source>Cancer Immunol Res</source>. (<year>2016</year>) <volume>4</volume>:<page-range>412&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/2326-6066.CIR-15-0240</pub-id>, PMID: <pub-id pub-id-type="pmid">26968205</pub-id></citation></ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Formenti</surname> <given-names>SC</given-names>
</name>
<name>
<surname>Rudqvist</surname> <given-names>NP</given-names>
</name>
<name>
<surname>Golden</surname> <given-names>E</given-names>
</name>
<name>
<surname>Cooper</surname> <given-names>B</given-names>
</name>
<name>
<surname>Wennerberg</surname> <given-names>E</given-names>
</name>
<name>
<surname>Lhuillier</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Radiotherapy induces responses of lung cancer to CTLA-4 blockade</article-title>. <source>Nat Med</source>. (<year>2018</year>) <volume>24</volume>:<page-range>1845&#x2013;51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41591-018-0232-2</pub-id>, PMID: <pub-id pub-id-type="pmid">30397353</pub-id></citation></ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Caushi</surname> <given-names>JX</given-names>
</name>
<name>
<surname>Asmar</surname> <given-names>M EI</given-names>
</name>
<name>
<surname>Anagnostou</surname> <given-names>V</given-names>
</name>
<name>
<surname>Cottrell</surname> <given-names>TR</given-names>
</name>
<etal/>
</person-group>. <article-title>Compartmental analysis of T-cell clonal dynamics as a function of pathologic response to neoadjuvant PD-1 blockade in resectable non-small cell lung cancer</article-title>. <source>Clin Cancer Res</source>. (<year>2020</year>) <volume>26</volume>:<page-range>1327&#x2013;37</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/1078-0432.CCR-19-2931</pub-id>, PMID: <pub-id pub-id-type="pmid">31754049</pub-id></citation></ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>DeWitt</surname> <given-names>WS</given-names>
</name>
<name>
<surname>Emerson</surname> <given-names>RO</given-names>
</name>
<name>
<surname>Lindau</surname> <given-names>P</given-names>
</name>
<name>
<surname>Vignail</surname> <given-names>M</given-names>
</name>
<name>
<surname>Snyder</surname> <given-names>TM</given-names>
</name>
<name>
<surname>Desmarais</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Dynamics of the cytotoxic T cell response to a model of acute viral infection</article-title>. <source>J Virol</source>. (<year>2015</year>) <volume>89</volume>:<page-range>4517&#x2013;26</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/JVI.03474-14</pub-id>, PMID: <pub-id pub-id-type="pmid">25653453</pub-id></citation></ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chu</surname> <given-names>ND</given-names>
</name>
<name>
<surname>Bi</surname> <given-names>HS</given-names>
</name>
<name>
<surname>Emerson</surname> <given-names>RO</given-names>
</name>
<name>
<surname>Sherwood</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Birnbaum</surname> <given-names>ME</given-names>
</name>
<name>
<surname>Robins</surname> <given-names>HS</given-names>
</name>
<etal/>
</person-group>. <article-title>Longitudinal immunosequencing in healthy people reveals persistent T cell receptors rich in highly public receptors</article-title>. <source>BMC Immunol</source>. (<year>2019</year>) <volume>20</volume>:<fpage>19</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12865-019-0300-5</pub-id>, PMID: <pub-id pub-id-type="pmid">31226930</pub-id></citation></ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kanakry</surname> <given-names>CG</given-names>
</name>
<name>
<surname>Coffey</surname> <given-names>DG</given-names>
</name>
<name>
<surname>Towlerton</surname> <given-names>AMH</given-names>
</name>
<name>
<surname>Vulic</surname> <given-names>A</given-names>
</name>
<name>
<surname>Storer</surname> <given-names>BE</given-names>
</name>
<name>
<surname>Chou</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Origin and evolution of the T cell repertoire after posttransplantation cyclophosphamide</article-title>. <source>JCI Insight</source>. (<year>2016</year>) <volume>1</volume>:<elocation-id>e86252</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1172/jci.insight.86252</pub-id>, PMID: <pub-id pub-id-type="pmid">27213183</pub-id></citation></ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Soto</surname> <given-names>C</given-names>
</name>
<name>
<surname>Bombardi</surname> <given-names>RG</given-names>
</name>
<name>
<surname>Kozhevnikov</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sinkovits</surname> <given-names>RS</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>EC</given-names>
</name>
<name>
<surname>Branchizio</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>High frequency of shared clonotypes in human T cell receptor repertoires</article-title>. <source>Cell Rep</source>. (<year>2020</year>) <volume>32</volume>:<fpage>107882</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.celrep.2020.107882</pub-id>, PMID: <pub-id pub-id-type="pmid">32668251</pub-id></citation></ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scheper</surname> <given-names>W</given-names>
</name>
<name>
<surname>Kelderman</surname> <given-names>S</given-names>
</name>
<name>
<surname>Fanchi</surname> <given-names>LF</given-names>
</name>
<name>
<surname>Linnemann</surname> <given-names>C</given-names>
</name>
<name>
<surname>Bendle</surname> <given-names>G</given-names>
</name>
<name>
<surname>de Rooij</surname> <given-names>MAJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Low and variable tumor reactivity of the intratumoral TCR repertoire in human cancers</article-title>. <source>Nat Med</source>. (<year>2019</year>) <volume>25</volume>:<fpage>89</fpage>&#x2013;<lpage>94</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41591-018-0266-5</pub-id>, PMID: <pub-id pub-id-type="pmid">30510250</pub-id></citation></ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>F</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>C</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Attention-emotion-enhanced convolutional LSTM for sentiment analysis</article-title>. <source>IEEE Trans Neural Netw Learn Syst</source>. (<year>2022</year>) <volume>33</volume>:<page-range>4332&#x2013;45</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TNNLS.2021.3056664</pub-id>, PMID: <pub-id pub-id-type="pmid">33600326</pub-id></citation></ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tickotsky</surname> <given-names>N</given-names>
</name>
<name>
<surname>Sagiv</surname> <given-names>T</given-names>
</name>
<name>
<surname>Prilusky</surname> <given-names>J</given-names>
</name>
<name>
<surname>Shifrut</surname> <given-names>E</given-names>
</name>
<name>
<surname>Friedman</surname> <given-names>N</given-names>
</name>
</person-group>. <article-title>McPAS-TCR: a manually curated catalogue of pathology-associated T cell receptor sequences</article-title>. <source>Bioinformatics</source>. (<year>2017</year>) <volume>33</volume>:<page-range>2924&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btx286</pub-id>, PMID: <pub-id pub-id-type="pmid">28481982</pub-id></citation></ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>SY</given-names>
</name>
<name>
<surname>Yue</surname> <given-names>T</given-names>
</name>
<name>
<surname>Lei</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>AY</given-names>
</name>
</person-group>. <article-title>TCRdb: a comprehensive database for T-cell receptor sequences with powerful search function</article-title>. <source>Nucleic Acids Res</source>. (<year>2021</year>) <volume>49</volume>:<fpage>D468</fpage>&#x2013;<lpage>D474</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkaa796</pub-id>, PMID: <pub-id pub-id-type="pmid">32990749</pub-id></citation></ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shugay</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bagaev</surname> <given-names>DV</given-names>
</name>
<name>
<surname>Zvyagin</surname> <given-names>IV</given-names>
</name>
<name>
<surname>Vroomans</surname> <given-names>RM</given-names>
</name>
</person-group>. <article-title>VDJdb: a curated database of T-cell receptor sequences with known antigen specificity</article-title>. <source>Nucleic Acids Res</source>. (<year>2018</year>) <volume>46</volume>:<page-range>D419&#x2013;27</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkx760</pub-id>, PMID: <pub-id pub-id-type="pmid">28977646</pub-id></citation></ref>
</ref-list>
</back>
</article>