<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2024.1345586</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>DeepLION2: deep multi-instance contrastive learning framework enhancing the prediction of cancer-associated T cell receptors by attention strategy on motifs</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Qian</surname>
<given-names>Xinyang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1647830"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Yang</surname>
<given-names>Guang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Fan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1841716"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Xuanping</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1821325"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhu</surname>
<given-names>Xiaoyan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1821368"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lai</surname>
<given-names>Xin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xiao</surname>
<given-names>Xiao</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Tao</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Jiayin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/615156"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Computer Science and Technology, Xi&#x2019;an Jiaotong University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Shaanxi Engineering Research Center of Medical and Health Big Data, Xi&#x2019;an Jiaotong University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Clinical Oncology, The Second Affiliated Hospital of Air Force Medical University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Genomics Institute</institution>, <addr-line>Geneplus-Shenzhen, Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Department of Thoracic Surgery, The Second Affiliated Hospital of Air Force Medical University</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Yi Shi, Shanghai Jiao Tong University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Scott Christley, University of Texas Southwestern Medical Center, United States</p>
<p>Chong Chu, Harvard Medical School, United States</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Jiayin Wang, <email xlink:href="mailto:wangjiayin@mail.xjtu.edu.cn">wangjiayin@mail.xjtu.edu.cn</email>; Tao Wang, <email xlink:href="mailto:tddocwangt@163.com">tddocwangt@163.com</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>03</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1345586</elocation-id>
<history>
<date date-type="received">
<day>28</day>
<month>11</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>19</day>
<month>02</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Qian, Yang, Li, Zhang, Zhu, Lai, Xiao, Wang and Wang</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Qian, Yang, Li, Zhang, Zhu, Lai, Xiao, Wang and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>T cell receptor (TCR) repertoires provide valuable insights into complex human diseases, including cancers. Recent advancements in immune sequencing technology have significantly improved our understanding of TCR repertoire. Some computational methods have been devised to identify cancer-associated TCRs and enable cancer detection using TCR sequencing data. However, the existing methods are often limited by their inadequate consideration of the correlations among TCRs within a repertoire, hindering the identification of crucial TCRs. Additionally, the sparsity of cancer-associated TCR distribution presents a challenge in accurate prediction.</p>
</sec>
<sec>
<title>Methods</title>
<p>To address these issues, we presented DeepLION2, an innovative deep multi-instance contrastive learning framework specifically designed to enhance cancer-associated TCR prediction. DeepLION2 leveraged content-based sparse self-attention, focusing on the top <italic>k</italic> related TCRs for each TCR, to effectively model inter-TCR correlations. Furthermore, it adopted a contrastive learning strategy for bootstrapping parameter updates of the attention matrix, preventing the model from fixating on non-cancer-associated TCRs.</p>
</sec>
<sec>
<title>Results</title>
<p>Extensive experimentation on diverse patient cohorts, encompassing over ten cancer types, demonstrated that DeepLION2 significantly outperformed current state-of-the-art methods in terms of accuracy, sensitivity, specificity, Matthews correlation coefficient, and area under the curve (AUC). Notably, DeepLION2 achieved impressive AUC values of 0.933, 0.880, and 0.763 on thyroid, lung, and gastrointestinal cancer cohorts, respectively. Furthermore, it effectively identified cancer-associated TCRs along with their key motifs, highlighting the amino acids that play a crucial role in TCR-peptide binding.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>These compelling results underscore DeepLION2's potential for enhancing cancer detection and facilitating personalized cancer immunotherapy. DeepLION2 is publicly available on GitHub, at <uri xlink:href="https://github.com/Bioinformatics7181/DeepLION2">https://github.com/Bioinformatics7181/DeepLION2</uri>, for academic use only.</p>
</sec>
</abstract>
<kwd-group>
<kwd>T cell receptor</kwd>
<kwd>TCR repertoire data analysis</kwd>
<kwd>cancer-associated TCR</kwd>
<kwd>machine learning approach</kwd>
<kwd>deep learning framework</kwd>
<kwd>multi-instance learning</kwd>
<kwd>sparse self-attention</kwd>
<kwd>contrastive learning</kwd>
</kwd-group>
<contract-sponsor id="cn001">Natural Science Basic Research Program of Shaanxi Province<named-content content-type="fundref-id">10.13039/501100017596</named-content>
</contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="4"/>
<equation-count count="20"/>
<ref-count count="43"/>
<page-count count="14"/>
<word-count count="9099"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Cancer Immunity and Immunotherapy</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>T cells are crucial elements in human immune system, capable of recognizing and responding to various antigens, including tumors, through their T cell receptors (TCRs) (<xref ref-type="bibr" rid="B1">1</xref>&#x2013;<xref ref-type="bibr" rid="B3">3</xref>). In cancers, specific TCRs with distinct characteristics emerge in patients&#x2019; T cell repertoire, referred to cancer-associated TCRs (caTCRs) (<xref ref-type="bibr" rid="B4">4</xref>). These caTCRs possess unique adaptations to interact with tumor-related antigens, contributing to the immune response against cancer. They also exhibit shared biochemical signatures among caTCRs targeting the same cancer type or subtype, holding promise for cancer detection and treatment (<xref ref-type="bibr" rid="B5">5</xref>&#x2013;<xref ref-type="bibr" rid="B8">8</xref>). Although the precise biochemical properties distinguishing caTCRs are still under exploration, advancements in the Adaptive Immune Receptor Repertoire sequencing (AIRR-seq) have revolutionized our understanding of TCR repertoires at both individual and population levels, generating vast sequencing data (<xref ref-type="bibr" rid="B9">9</xref>). Consequently, computational frameworks have been developed to predict caTCRs, some to differentiate cancer-associated repertoires from non-cancer ones (<xref ref-type="bibr" rid="B10">10</xref>&#x2013;<xref ref-type="bibr" rid="B18">18</xref>). By leveraging the data obtained through AIRR-seq, these computational frameworks play a crucial role in early cancer screening and the prediction of cancer immune responses and immunotherapy effectiveness (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B19">19</xref>). Moreover, they contribute to detecting molecular residual diseases and interpreting tumor mutation burdens, serving as pivotal biomarkers for assessing a patient&#x2019;s prognostic status (<xref ref-type="bibr" rid="B20">20</xref>&#x2013;<xref ref-type="bibr" rid="B22">22</xref>).</p>
<p>Predicting caTCRs from AIRR-seq data has been defined as a multi-instance learning (MIL) task, with individual TCRs as &#x2018;instances&#x2019; and the entire repertoire as the &#x2018;bag&#x2019; (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B18">18</xref>, <xref ref-type="bibr" rid="B23">23</xref>). Current computational methods predominantly focus on the complementarity determining region 3 (CDR3) of the TCR&#x3b2; chain, involving two crucial components: CDR3 sequence feature extraction and the application of MIL techniques. Regarding sequence feature extraction, traditional methods, based on similarity comparisons of entire sequences, faced challenges in pinpointing specific amino acid residues, or &#x2018;motifs,&#x2019; crucial for antigen recognition (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B13">13</xref>). To address this limitation, some researchers preprocessed sequences into fixed-length overlapping fragments (<xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B14">14</xref>). However, motif lengths remained variable, constraining their performance (<xref ref-type="bibr" rid="B14">14</xref>). DeepLION first designed the model that enables to accommodate motifs of various lengths, surpassing existing methods in feature extraction (<xref ref-type="bibr" rid="B17">17</xref>). In the context of applying MIL techniques, early methods primarily considered the most significant CDR3 sequence, neglecting other valuable sequences (<xref ref-type="bibr" rid="B10">10</xref>&#x2013;<xref ref-type="bibr" rid="B15">15</xref>). To address this issue, DeepTCR employed a multi-head attention mechanism, while DeepLION used a linear classifier, assigning appropriate weights to CDR3 sequences in the repertoire (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B17">17</xref>). To further tackle the issue of a small fraction of caTCRs within the repertoire, MINN_SA applied a sparsity constraint to the linear classifier&#x2019;s output, focusing attention on the caTCRs within the repertoire, and achieved superior performance compared to popular MIL methods (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B18">18</xref>).</p>
<p>Unfortunately, in the application of MIL techniques, there are still two key issues that prevent accurate predictions from the existing methods: their inadequate consideration of the correlations among the TCRs within a repertoire, and the sparsity of caTCR distribution. On one hand, TCRs with similar or even identical CDR3 sequences can recognize different antigens based on their distinct structural characteristics (<xref ref-type="bibr" rid="B24">24</xref>). Consequently, those sequence-based methods are susceptible to misclassification in such cases, including mislabeling non-cancer TCRs with similar or identical sequences to caTCRs as caTCRs, resulting in false positives. Fortunately, utilizing the context of TCRs, specifically calculating the correlations between TCRs, enhances the inference of TCR antigen-binding specificity and enables an accurate caTCR prediction (<xref ref-type="bibr" rid="B25">25</xref>). While TransMIL effectively integrated the self-attention mechanism within its MIL component to capture inter-instance correlations, showing impressive performance in whole slide image classification (<xref ref-type="bibr" rid="B26">26</xref>), dedicated methods for caTCR prediction remain to be developed. On the other hand, tumor-infiltrating lymphocyte repertoires often contain over 80% of TCRs lacking tumor reactivity, indicating the sparse distribution of caTCRs (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B27">27</xref>). In such cases, the self-attention mechanism calculates a group of attention scores for a TCR compared to all others, which may inadvertently allocate excessive attention to unrelated TCRs, and further generate erroneous predictions. In addition, insufficient samples from patients with the same cancer type may impede the model&#x2019;s ability to focus on the sparse caTCRs, limiting the prediction performance of the self-attention mechanism (<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B26">26</xref>).</p>
<p>In summary, the current computational methods for caTCR prediction are constrained by their limited consideration of the correlations among TCRs in the repertoire and the sparsity of caTCR distribution. To address these issues, we proposed a novel MIL method called DeepLION2, which incorporated sparse self-attention and contrastive learning, to enhance the prediction of caTCRs using TCR sequencing (TCR-seq) data. It met the requirement to consider TCR correlations and to identify sparse caTCRs within the repertoire by utilizing a content-based sparse attention mechanism. This mechanism focused only on the <italic>k</italic> most relevant TCRs for each TCR, avoiding unnecessary attention on unrelated TCRs. Additionally, we integrated a self-contrastive learning strategy into model training to enhance the attention matrix by focusing on sparse caTCRs and thereby improving caTCR prediction. In our nested cross-validation evaluation, DeepLION2 outperformed the state-of-the-art methods in caTCR prediction and repertoire classification and achieved impressive area under the receiver operating characteristic (ROC) curve (AUC) values of 0.933, 0.880, and 0.763 on raw TCR-seq data of thyroid cancer (THCA), lung cancer (LUCA), and gastrointestinal cancer (GICA) patient cohorts, respectively. Moreover, it could effectively identify caTCRs along with their key motifs, which are essential for TCR-peptide binding. These results highlight its potential to advance cancer research and facilitate personalized cancer immunotherapy.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<p>DeepLION2 set up a three-component workflow, comprising data preprocessing, TCR antigen-specificity extraction, and MIL, which closely resembled the workflow of DeepLION (<xref ref-type="bibr" rid="B17">17</xref>) (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>). While the initial two parts of DeepLION2 employed the same methodology as DeepLION for data preprocessing and TCR antigen-specificity extraction, the main improvement was observed in the MIL component. In the third part, DeepLION2 introduced a content-based sparse self-attention mechanism in conjunction with contrastive learning to effectively aggregate TCR features and embed the repertoire. It performed both self-attention and sparse self-attention computations and compared the results to optimize attention learning. By considering the relationships among TCRs within the repertoire and the sparsity of caTCRs, it significantly enhanced the aggregation process, enabling accurate prediction of whether the TCR repertoire was cancerous or non-cancerous.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>DeepLION2 for accurate prediction of cancer-associated TCRs. <bold>(A)</bold> The workflow of DeepLION2 contained three parts: data preprocessing, TCR antigen-specificity extraction, and multi-instance learning. In the data preprocessing, raw sequences of length <italic>l</italic> were embedded into the TCR matrix with dimension <italic>l</italic> &#xd7; 15 after sequence filtering. Then the antigen-specificity of each TCR was extracted by a convolutional network and the corresponding feature was generated. In the last part, DeepLION2 used a content-based sparse self-attention to capture the correlations between each TCR and its top <italic>k</italic> related TCRs. Moreover, it also performed self-attention calculation for self-contrastive learning, where the outputs of sparse self-attention and self-attention were compared to improve attention learning. Finally, the attention output was linearly mapped and pooled to generate the cancer score for the repertoire. <bold>(B)</bold> The computational details of content-based sparse self-attention used in DeepLION2. First, the TCR-to-TCR affinity graph <inline-formula>
<mml:math display="inline" id="im1">
<mml:mi>A</mml:mi>
</mml:math>
</inline-formula> was calculated with <italic>Q</italic> and <italic>K</italic>, measuring the correlation between TCRs. Then the index matrix <inline-formula>
<mml:math display="inline" id="im2">
<mml:mi>I</mml:mi>
</mml:math>
</inline-formula> was derived with the row-wise extraction operation, which recorded the <italic>k</italic> indices of the related TCRs for each TCR. In the subsequent computation of self-attention, the <italic>k</italic> most relevant TCRs for each TCR were selectively considered, thereby mitigating the impact of less relevant TCRs. Finally, the output <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was obtained after the computation of self-attention. MM, matrix multiplication; T, transposition.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1345586-g001.tif"/>
</fig>
<sec id="s2_1">
<label>2.1</label>
<title>Data preprocessing</title>
<p>To effectively utilize TCR-seq data for caTCR prediction and repertoire classification, preprocessing steps are necessary, involving sequence filtering and embedding. Considering that existing studies on TCRs predominantly focused on the &#x3b2; chain, we only kept the &#x3b2; chain CDR3 sequences as the input of DeepLION2. During sequence filtering, low-quality CDR3 sequences and those unrelated to cancer were removed. As described in the previous studies (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B17">17</xref>), the following types of sequences were removed: I. sequences with inadequate length (&lt; 10) or excessive length (&gt; 24), II. sequences featuring special characters (X, +, *, etc.), III. incomplete sequences, not commencing with cysteine (C) or culminating with phenylalanine (F), IV. sequences with an unresolved variable gene locus, and V. sequences appearing in the reference dataset from Xu&#x2019;s study, frequently observed in both healthy individuals and cancer patients. From the remaining sequences, the <italic>N</italic> sequences with the highest abundance were selected. In the sequence embedding process, sequences were encoded into numerical matrices that effectively contained the antigen-binding specificity of the CDR3 for downstream analysis. A sequence of length <italic>l</italic> could be encoded into an <italic>l</italic> &#xd7; <italic>d</italic> TCR matrix using a 20 &#xd7; <italic>d</italic> feature matrix, where each of the 20 amino acids was represented by a feature vector of dimension <italic>d</italic>. Among popular feature matrices (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B28">28</xref>, <xref ref-type="bibr" rid="B29">29</xref>), given that the Beshnova matrix contains more biochemical information and has demonstrated good performance in methods like DeepLION and DeepCAT, our method also adopted it for encoding sequences of length <italic>l</italic> into an <italic>l</italic> &#xd7; 15 TCR matrix.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>TCR antigen-specificity extraction considering the cancer-associated motifs of different lengths</title>
<p>Properly extracting TCR antigen-specificity is essential for identifying caTCRs, and various computational methods have been used for this purpose. Compared with other methods, DeepLION introduced a convolution network with convolutional filters of different sizes to consider the key motifs of different lengths in TCRs, resulting in improved performance (<xref ref-type="bibr" rid="B17">17</xref>). As a result, DeepLION2 adopted a similar network architecture to DeepLION to extract TCR antigen-specificity. It consisted of 14 convolutional filters with different sizes that performed convolution operations on the TCR matrix, generating corresponding convolution mappings. The 1-max pooling function was then applied to reduce each mapping dimension to 1. By concatenating these mapping results, a feature with a dimension of 1 &#xd7; 14 representing the TCR antigen-specificity was obtained for each TCR. Finally, this process resulted in an <italic>N</italic> &#xd7; 14 matrix for each TCR repertoire.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Multi-instance learning properly modeling the relationships among TCRs</title>
<sec id="s2_3_1">
<label>2.3.1</label>
<title>Self-attention for calculating the correlation scores between TCRs</title>
<p>Considering the relationships among TCRs within the repertoire can better estimate the antigen-binding specificity of TCRs with similar/identical CDR3 sequences. Self-attention-based models have proven to be quite effective in processing sequence data and calculating relationship scores between instances in MIL. Similar to these models, the self-attention mechanism in DeepLION2 was formally defined as <xref ref-type="disp-formula" rid="eq1">Equations (1</xref>, <xref ref-type="disp-formula" rid="eq2">2)</xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:msup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>q</mml:mi>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mi>q</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:msup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:msup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>v</mml:mi>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mi>v</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>Attention</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">V</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mtext>softmax</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:msup>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mtext>T</mml:mtext>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> was an input with <italic>N</italic> instances, whose dimension was <italic>D</italic> (<italic>D</italic> = 14 in DeepLION2 due to the TCR feature matrix), <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>K</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> were query, key and value matrices obtained by linear transformation of <italic>X</italic>, where <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>'</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> was the dimension of the instance after transformation, and softmax(&#xb7;) was the activation function to normalize the results and get the final output <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The output of <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:mtext>softmax</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="italic">QK</mml:mi>
<mml:mtext>T</mml:mtext>
</mml:msup>
<mml:mo stretchy="false">/</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>'</mml:mo>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> could be seen as the attention score matrix computed, containing the relationship scores between the corresponding instances. To avoid weight concentration and gradient vanishing, the scalar factor <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>'</mml:mo>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</inline-formula> was introduced (<xref ref-type="bibr" rid="B25">25</xref>).</p>
<p>To improve the self-attention mechanism&#x2019;s performance, a preferred method is multi-head self-attention, formally defined as <xref ref-type="disp-formula" rid="eq3">Equations (3</xref>&#x2013;<xref ref-type="disp-formula" rid="eq5">5)</xref>:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>Attention</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mrow>
<mml:mi>MHSA</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>MHSA</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mtext>Concat</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">H</mml:mi>
<mml:mi mathvariant="bold-italic">e</mml:mi>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold-italic">d</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the multi-head self-attention computation process, <italic>Q</italic>, <italic>K</italic> and <italic>V</italic> were divided into <italic>h</italic> equal parts, where <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> were used to computed the <italic>i</italic>th attention head <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. And then the outputs of <italic>h</italic> heads were concatenated as the final attention output <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. To facilitate the stacking of self-attention blocks, the output <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was linearly transformed to the output <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>'</mml:mo>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, whose dimension was the same as the input <italic>X</italic> as <xref ref-type="disp-formula" rid="eq6">Equation (6)</xref>:</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>'</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>A</mml:mi>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mi>A</mml:mi>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the context of caTCR prediction, the input <italic>X</italic> of the self-attention was the <italic>N</italic> &#xd7; 14 TCR repertoire feature matrix, where each row represented a TCR and <italic>N</italic> represented the total number of TCRs. This mechanism calculated correlation scores between TCRs, enabling the accurate extraction of cancer-associated biochemical features and precise identification of caTCRs within the repertoire.</p>
</sec>
<sec id="s2_3_2">
<label>2.3.2</label>
<title>Content-based sparse self-attention prioritizing top <italic>k</italic> related TCRs for each TCR</title>
<p>Due to the sparsity of caTCRs, calculating relationship scores between all TCRs may inadvertently shift the focus toward non-cancer-associated TCRs, thereby potentially reducing caTCR prediction performance. To tackle this, the preferred solution is sparse self-attention, which falls into two categories: position-based and content-based approaches (<xref ref-type="bibr" rid="B30">30</xref>, <xref ref-type="bibr" rid="B31">31</xref>). Position-based attention restricts the attention matrix based on predefined position-related patterns, but the distribution pattern of caTCRs in the repertoire remains unknown. As a result, DeepLION2 incorporated the content-based sparse attention to only consider the top <italic>k</italic> related TCRs for each TCR in the repertoire (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>). The formal definition was as <xref ref-type="disp-formula" rid="eq7">Equations (7</xref>&#x2013;<xref ref-type="disp-formula" rid="eq11">11)</xref>:</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:msup>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mtext>T</mml:mtext>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mo>=</mml:mo>
<mml:mtext>topkIndex</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mi>u</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mtext>unsqueeze</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msup>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi>g</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mtext>gather</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">I</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mi>g</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mtext>gather</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">I</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>CSA</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mtext>squeeze</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext>Attention</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mi>u</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi>g</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mi>g</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>'</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>A</mml:mi>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mi>A</mml:mi>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>First, we derived the TCR-to-TCR affinity graph <inline-formula>
<mml:math display="inline" id="im19">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> via matrix multiplication between <italic>Q</italic> and transposed <italic>K</italic>, where measured the correlation of each TCR with other TCRs. And then we pruned the affinity graph <inline-formula>
<mml:math display="inline" id="im20">
<mml:mi>A</mml:mi>
</mml:math>
</inline-formula> by retaining only the first <italic>k</italic> connections of each TCR based on the values of the elements in <inline-formula>
<mml:math display="inline" id="im21">
<mml:mi>A</mml:mi>
</mml:math>
</inline-formula> (i.e., the correlation between TCRs). The index matrix <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> was derived with the row-wise extraction operation, which recorded the <italic>k</italic> indices of the related TCRs for each TCR. Specifically, the <italic>i</italic>th row of <italic>I</italic> included <italic>k</italic> indices of the most relevant TCRs for the <italic>i</italic>th TCR. The gathered key and value matrices <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msup>
<mml:mi>K</mml:mi>
<mml:mi>g</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>g</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, which contained the top <italic>k</italic> TCR vectors for each TCR, were obtained by gathering <italic>K</italic> and <italic>V</italic> with <italic>I</italic> (i.e., extracting the corresponding elements according to the indices in <italic>I</italic>). For facilitating the following matrix multiplication computations, the unsqueezed query matrix <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mi>u</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> was obtained by ascending the dimension of <italic>Q</italic>. Finally, the self-attention was applied on the <italic>Q<sup>u</sup>
</italic>, <italic>K<sup>g</sup>
</italic> and <italic>V<sup>g</sup>
</italic>, and the output <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> was obtained after the squeeze operation. Consistent with the multi-head self-attention, we obtained the final output <inline-formula>
<mml:math display="inline" id="im27">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>'</mml:mo>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> after the linear transformation. In conclusion, by precomputing TCR correlations before applying traditional self-attention, DeepLION2 selectively considered the <italic>k</italic> most relevant TCRs for each TCR, thereby mitigating the impact of less relevant TCRs.</p>
</sec>
<sec id="s2_3_3">
<label>2.3.3</label>
<title>Self-contrastive learning for robust attention learning</title>
<p>Content-based sparse self-attention has shown excellent performance on classification tasks when sufficient training data (&gt; one thousand samples) is available (<xref ref-type="bibr" rid="B31">31</xref>). However, obtaining TCR-seq data from a sufficient number of patients with the same cancer type is challenging, and if there isn&#x2019;t enough training data, the model struggles to focus on the sparse caTCRs in the repertoire. To address this challenge, DeepLION2 incorporated self-contrastive learning in its MIL component. We first assumed that each TCR within a repertoire exclusively relates to others recognizing the same antigen, signifying an attention score of 0 with unrelated TCRs. Based on this assumption, we inferred that the output, despite lacking specific constraints in self-attention calculation, would be identical to that derived from sparse self-attention. To ensure this, we performed both self-attention and sparse self-attention calculations and compared their outputs using the mean square error loss function, in the aim to minimize the discrepancy between the two outputs during the model training. The loss function was defined as <xref ref-type="disp-formula" rid="eq12">Equation (12)</xref>:</p>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x2112;</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>MSE</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>'</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>'</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>D</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>o</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover> <mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im28">
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im29">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>o</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> were the elements of the <italic>i</italic>-th row and <italic>j</italic>-th column in the output matrix <inline-formula>
<mml:math display="inline" id="im30">
<mml:mrow>
<mml:msubsup>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>'</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im31">
<mml:mrow>
<mml:msubsup>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>'</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. In the model training process, by optimizing the loss function <inline-formula>
<mml:math display="inline" id="im32">
<mml:mrow>
<mml:msub>
<mml:mi>&#x2112;</mml:mi>
<mml:mtext>C</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the attention score matrix of the sparse self-attention was constantly revised, where each TCR focused only on relevant TCRs and ignored other TCRs. Consequently, this strategy allowed DeepLION2 to focus more on the sparse caTCRs within a repertoire with small sample sizes.</p>
</sec>
<sec id="s2_3_4">
<label>2.3.4</label>
<title>Decision layer for making prediction both for TCRs and repertoires</title>
<p>The decision layer was designed to make the final predictions for individual TCRs and the TCR repertoire. The output of the content-based sparse self-attention <inline-formula>
<mml:math display="inline" id="im33">
<mml:mrow>
<mml:msubsup>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>'</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> was linearly transformed to integrate the features of each TCR, as <xref ref-type="disp-formula" rid="eq13">Equation (13)</xref>:</p>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mtext>Sigmoid</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo>'</mml:mo>
</mml:msubsup>
<mml:msup>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mi>D</mml:mi>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mi>D</mml:mi>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im34">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>D</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im35">
<mml:mrow>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mi>D</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mi>D</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> were the weight and the bias of the linear transformation, Sigmoid(&#xb7;) was the activation function to map the values to the interval (0, 1), and <inline-formula>
<mml:math display="inline" id="im36">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represented the prediction scores of TCRs in the repertoire. And then the average pooling was used to mapping the <inline-formula>
<mml:math display="inline" id="im37">
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> into the TCR repertoire prediction result as <xref ref-type="disp-formula" rid="eq14">Equation (14)</xref>:</p>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>|</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im38">
<mml:mrow>
<mml:mtext>P</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>|</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denoted the probability that the TCR repertoire is associated with cancer (i.e., the probability that a patient has cancer). When <inline-formula>
<mml:math display="inline" id="im39">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mo>&gt;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, the repertoire was predicted to be cancer-associated, and to be noncancerous otherwise.</p>
<p>The whole model DeepLION2 was end-to-end trainable, and the loss function <inline-formula>
<mml:math display="inline" id="im40">
<mml:mi>&#x2112;</mml:mi>
</mml:math>
</inline-formula> used for model training was defined as <xref ref-type="disp-formula" rid="eq15">Equations (15</xref>, <xref ref-type="disp-formula" rid="eq16">16)</xref>:</p>
<disp-formula id="eq15">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x2112;</mml:mi>
<mml:mi>M</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>CE</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mo>,</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>log</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xb7;</mml:mo>
<mml:mtext>log</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq16">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x2112;</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x2112;</mml:mi>
<mml:mi>M</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x2112;</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im41">
<mml:mi>&#x2112;</mml:mi>
</mml:math>
</inline-formula> consisted of the main loss function <inline-formula>
<mml:math display="inline" id="im42">
<mml:mrow>
<mml:msub>
<mml:mi>&#x2112;</mml:mi>
<mml:mtext>M</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the self-contrastive learning loss function <inline-formula>
<mml:math display="inline" id="im43">
<mml:mrow>
<mml:msub>
<mml:mi>&#x2112;</mml:mi>
<mml:mtext>C</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. By optimizing the log-likelihood function <inline-formula>
<mml:math display="inline" id="im44">
<mml:mrow>
<mml:msub>
<mml:mi>&#x2112;</mml:mi>
<mml:mtext>M</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, DeepLION2 learned to predict whether a repertoire is cancer-associated. Additionally, a constraint term <inline-formula>
<mml:math display="inline" id="im45">
<mml:mrow>
<mml:msub>
<mml:mi>&#x2112;</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was optimized to enhance the learning of sparse self-attention and improve prediction performance.</p>
<p>The trained DeepLION2 not only predicted the cancer status of patient samples but also identified caTCRs within a repertoire through the TCR score vector <inline-formula>
<mml:math display="inline" id="im46">
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>. Each element <inline-formula>
<mml:math display="inline" id="im47">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in the vector represented the probability that the <italic>i</italic>th TCR in the repertoire is a caTCR. In a predicted cancerous repertoire, the probability <inline-formula>
<mml:math display="inline" id="im48">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> served as a reliable indicator: the higher the probability, the stronger the likelihood that the corresponding TCR is associated with cancer. Conversely, regardless of their respective probabilities, every TCR in a predicted noncancerous repertoire is unassociated with cancer. Additionally, DeepLION2 could also identify the key motifs of caTCRs by calculating the motif scores according to the weight parameters of the trained model.</p>
</sec>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<p>To evaluate the performance of DeepLION2, we conducted experiments on diverse cohorts of patients with various cancer types. We first described in detail the experimental data, comparison models, evaluation metrics, and cross-validation approach. Then we specifically assessed the enhancement of the MIL component of DeepLION2 using preprocessed real data. Furthermore, we applied DeepLION2 to raw TCR-seq data to assess its performance when predicting the caTCRs and repertoires. Finally, we demonstrated the key TCRs with their motifs from the raw data based on the trained models.</p>
<sec id="s3_1">
<label>3.1</label>
<title>Collecting data</title>
<p>We utilized two real datasets encompassing more than 10 cancer types for our experiments. The first dataset was obtained from The Cancer Genome Atlas (TCGA) database, comprising paired tumor and normal tissue samples from patients with ten distinct cancer types (<xref ref-type="bibr" rid="B32">32</xref>). These samples underwent preprocessing steps, including next-generation sequencing, TCR reconstruction techniques, and TCR encoding algorithms, as outlined in previous studies (<xref ref-type="bibr" rid="B33">33</xref>). Xiong et&#xa0;al. utilized this dataset in their review to evaluate the performance of existing MIL methods in cancer detection tasks, while Kim et&#xa0;al. also employed it for comparisons with other methods (<xref ref-type="bibr" rid="B18">18</xref>). Therefore, we employed this dataset to evaluate the MIL component of DeepLION2.</p>
<p>The second dataset was collected from the clinical database of Geneplus Technology Ltd. in Shenzhen (Geneplus) (<xref ref-type="bibr" rid="B34">34</xref>&#x2013;<xref ref-type="bibr" rid="B36">36</xref>). It consisted of raw TCR-seq data samples, including peripheral blood mononuclear cell and tumor-infiltrating lymphocyte samples from patients with THCA, LUCA, and GICA. Additionally, non-cancer individual peripheral blood mononuclear cell samples were included as the control cohorts. The LUCA samples encompassed samples of two cancer subtypes: lung squamous cell carcinoma (LUSC) and lung adenocarcinoma (LUAD), while the GICA samples were from patients of esophageal, gastric, colorectal, hepatocellular, and pancreatic cancers. Detailed information regarding the experimental data can be found in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>The details of the data used in the experiments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Dataset</th>
<th valign="middle" align="left">Disease</th>
<th valign="middle" align="left">Disease size</th>
<th valign="middle" align="left">Control size</th>
<th valign="middle" align="left">Total size</th>
<th valign="middle" align="left">Data source</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="10" align="left">TCGA</td>
<td valign="middle" align="left">BRCA</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">404</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B33">33</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">DLBC</td>
<td valign="middle" align="left">45</td>
<td valign="middle" align="left">45</td>
<td valign="middle" align="left">90</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B33">33</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">ESCA</td>
<td valign="middle" align="left">166</td>
<td valign="middle" align="left">166</td>
<td valign="middle" align="left">332</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B33">33</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">KIRC</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">404</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B33">33</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">LUAD</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">404</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B33">33</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">LUSC</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">404</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B33">33</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">OV</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">404</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B33">33</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">SKCM</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">404</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B33">33</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">STAD</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">202</td>
<td valign="middle" align="left">404</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B33">33</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">THYM</td>
<td valign="middle" align="left">108</td>
<td valign="middle" align="left">108</td>
<td valign="middle" align="left">216</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B33">33</xref>)</td>
</tr>
<tr>
<td valign="middle" rowspan="3" align="left">Geneplus</td>
<td valign="middle" align="left">THCA</td>
<td valign="middle" align="left">170</td>
<td valign="middle" align="left">260</td>
<td valign="middle" align="left">430</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B34">34</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">LUCA</td>
<td valign="middle" align="left">184</td>
<td valign="middle" align="left">260</td>
<td valign="middle" align="left">444</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B36">36</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">GICA</td>
<td valign="middle" align="left">151</td>
<td valign="middle" align="left">260</td>
<td valign="middle" align="left">411</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B35">35</xref>)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>TCGA, The Cancer Genome Atlas; BRCA, breast invasive carcinoma; DLBC, lymphoid neoplasm diffuse large B-cell lymphoma; ESCA, esophageal carcinoma; KIRC, kidney renal clear cell carcinoma; LUAD, lung adenocarcinoma; LUSC, lung squamous cell carcinoma; OV, ovarian serous cystadenocarcinoma; SKCM, skin cutaneous melanoma; STAD, stomach adenocarcinoma; THYM, thymoma; THCA, thyroid cancer; LUCA, lung cancer; GICA, gastrointestinal cancer.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Comparison model settings, evaluation metrics and validation approaches</title>
<p>To assess the enhancements introduced by DeepLION2, we compared it with several state-of-the-art methods: DeepTCR (<xref ref-type="bibr" rid="B16">16</xref>), DeepLION (<xref ref-type="bibr" rid="B17">17</xref>), MINN_SA (<xref ref-type="bibr" rid="B18">18</xref>), TransMIL (<xref ref-type="bibr" rid="B26">26</xref>), and BiFormer (<xref ref-type="bibr" rid="B31">31</xref>) (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). DeepTCR, DeepLION, and MINN_SA are embedded-space MIL methods specifically designed for TCR prediction. Due to their MIL design, DeepTCR is widely used in TCR studies, whereas DeepLION has been demonstrated to outperform earlier caTCR prediction methods, including DeepCAT (<xref ref-type="bibr" rid="B15">15</xref>). MINN_SA further took into account the sparsity of caTCRs and utilized sparse attention to selectively focus on the key TCRs within the repertoire while disregarding others, which has proven to perform better than popular existing MIL methods in this task. By contrast, TransMIL and BiFormer are not specific for TCR prediction. TransMIL employed the self-attention mechanism to consider inter-instance correlations and achieved significant improvement in whole slide image classification. BiFormer, a recent content-based sparse attention method in the field of computer vision, introduced bi-level routing attention and achieved higher classification accuracy than other self-attention-based methods. In order to ensure a fair comparison, we modified their network to predict TCRs by utilizing the same TCR feature extraction component as DeepLION2 and keeping only one layer as the MIL component.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Summary of the comparison models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Model</th>
<th valign="middle" align="left">Specific for TCR prediction?</th>
<th valign="middle" align="left">Considering correlations among instances?</th>
<th valign="middle" align="left">Considering the sparsity of instance distribution?</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">DeepTCR</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#xd7;</td>
</tr>
<tr>
<td valign="middle" align="left">DeepLION</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#xd7;</td>
</tr>
<tr>
<td valign="middle" align="left">MINN_SA</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#x221a;</td>
</tr>
<tr>
<td valign="middle" align="left">TransMIL</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#xd7;</td>
</tr>
<tr>
<td valign="middle" align="left">BiFormer</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#x221a;</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The hyperparameters of DeepLION2 contained the number of selected TCRs in date preprocessing <italic>N</italic>, the dimension <inline-formula>
<mml:math display="inline" id="im49">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>'</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and the head number <italic>h</italic> of self-attention/sparse self-attention, the ratio of sparse self-attention <italic>k<sub>r</sub>
</italic> (<italic>k<sub>r</sub>
</italic> = <italic>k/N</italic>), as well as the learning rate <italic>l<sub>r</sub>
</italic> and the epoch number <italic>e</italic> of model training. In the experiments, alignment with DeepLION, <italic>N</italic>, <italic>l<sub>r</sub>
</italic>, and <italic>e</italic> were set to 100, 0.001, and 500, whereas <inline-formula>
<mml:math display="inline" id="im50">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>'</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, <italic>h</italic>, and <italic>k<sub>r</sub>
</italic> were set to 10, 1, and 0.05 for low computational cost. For the comparison methods, we utilized the default hyperparameters, and DeepTCR only accepted TCR&#x3b2; sequences as input.</p>
<p>To assess the performance of DeepLION2 and the comparison models, we employed commonly used performance metrics in machine learning and statistical analysis within the biomedical field. These metrics included accuracy (ACC), sensitivity (SEN), specificity (SPE), Matthews correlation coefficient (MCC), and AUC. ACC, SEN, SPE, and MCC could be formally defined as <xref ref-type="disp-formula" rid="eq17">Equations (17</xref>&#x2013;<xref ref-type="disp-formula" rid="eq20">20)</xref>:</p>
<disp-formula id="eq17">
<label>(17)</label>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>ACC</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq18">
<label>(18)</label>
<mml:math display="block" id="M18">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>SEN</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq19">
<label>(19)</label>
<mml:math display="block" id="M19">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>SPE</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq20">
<label>(20)</label>
<mml:math display="block" id="M20">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>MCC</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext mathvariant="italic">TP</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext mathvariant="italic">TN</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext mathvariant="italic">FP</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext mathvariant="italic">FN</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="italic">TP</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext mathvariant="italic">FP</mml:mtext>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="italic">TP</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext mathvariant="italic">FN</mml:mtext>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="italic">TN</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext mathvariant="italic">FP</mml:mtext>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext mathvariant="italic">TN</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext mathvariant="italic">FN</mml:mtext>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>TP</italic> is the number of correct predictions in the positive samples, whereas <italic>FN</italic> is the number of wrong predictions in the positive samples, and <italic>TN</italic> is the number of correct predictions in the negative samples, whereas <italic>FP</italic> is the number of wrong predictions in the negative samples. Among the metrics, MCC is a correlation coefficient that quantifies the relationship between the true class and the prediction results, ranging from -1 to 1.</p>
<p>
<italic>K</italic>-fold cross-validation is a widely used validation approach for assessing model generalization. In <italic>K</italic>-fold cross-validation, the dataset was randomly divided into <italic>K</italic> equal parts, and <italic>K</italic> validations were performed, with each part serving as the test data while the remaining parts were used for training. This process ensures that the entire dataset is evaluated, providing an average performance estimation for the model. Unfortunately, it has been reported that <italic>K</italic>-fold cross-validation may yield skewed performance estimates when dealing with small sample sizes (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B37">37</xref>, <xref ref-type="bibr" rid="B38">38</xref>). To overcome this limitation, a refined technique called <italic>K</italic>-<italic>K&#x2019;</italic>-fold nested cross-validation was introduced. This enhanced approach aims to generate robust and unbiased performance estimates, irrespective of dataset size. In each of the <italic>K</italic> validations within the nested cross-validation, the training data was further divided into <italic>K&#x2019;</italic> equal parts, and <italic>K&#x2019;</italic>-fold cross-validation was performed to select the final models. As a result, considering the small sample size of the dataset used in the experiments, a 5-4-fold nested cross-validation approach was adopted to validate the performance of the models in our experiments.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Sparse self-attention and contrastive learning improves multi-instance learning for prediction of cancer-associated TCRs and repertoires</title>
<p>DeepLION2 concentrated on optimizing the MIL component to raise caTCR prediction accuracy. To achieve this improvement, content-based sparse self-attention was added, which helps to find important TCRs in the repertoire. Furthermore, the quality of attention-based learning outcomes was improved through the use of self-contrastive learning. To validate the effectiveness of these improvements, we compared the MIL component of DeepLION2 with those of other models using the TCGA dataset. Given that the relationship among TCRs and caTCR sparsity is not considered by either DeepTCR or DeepLION, we selected only DeepLION as a representative. For a fair comparison, we directly tested the models using 5-4-fold nested cross-validation on preprocessed samples from 10 different cancer types, without any additional processing. The validation results for all models across the ten cancer types were analyzed in terms of AUC and visualized in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>. Additionally, the results including all metrics can be found in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;1</bold>
</xref>. To facilitate comparison, the mean validation results for all metrics, across the ten cancer types, were summarized in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. These results could provide a comprehensive overview of the performance of the models, allowing for a detailed assessment of their effectiveness in predicting caTCRs.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>The AUC results of models on 10 cancer type samples from TCGA. BRCA, breast invasive carcinoma; DLBC, lymphoid neoplasm diffuse large B-cell lymphoma; ESCA, esophageal carcinoma; KIRC, kidney renal clear cell carcinoma; LUAD, lung adenocarcinoma; LUSC, lung squamous cell carcinoma; OV, ovarian serous cystadenocarcinoma; SKCM, skin cutaneous melanoma; STAD, stomach adenocarcinoma; THYM, thymoma; AUC, area under the receiver operating characteristic curve.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1345586-g002.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Mean results of models across 10 cancer type samples from TCGA.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left"/>
<th valign="middle" align="left">DeepLION</th>
<th valign="middle" align="left">MINN_SA</th>
<th valign="middle" align="left">TransMIL</th>
<th valign="middle" align="left">BiFormer</th>
<th valign="middle" align="left">DeepLION2</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">ACC</td>
<td valign="middle" align="left">0.631 &#xb1; 0.066</td>
<td valign="middle" align="left">0.629 &#xb1; 0.054</td>
<td valign="middle" align="left">0.655 &#xb1; 0.058</td>
<td valign="middle" align="left">0.627 &#xb1; 0.067</td>
<td valign="middle" align="left">
<bold>0.673 &#xb1; 0.057</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">SEN</td>
<td valign="middle" align="left">0.592 &#xb1; 0.112</td>
<td valign="middle" align="left">
<bold>0.672 &#xb1; 0.117</bold>
</td>
<td valign="middle" align="left">0.585 &#xb1; 0.130</td>
<td valign="middle" align="left">0.580 &#xb1; 0.129</td>
<td valign="middle" align="left">0.596 &#xb1; 0.139</td>
</tr>
<tr>
<td valign="middle" align="left">SPE</td>
<td valign="middle" align="left">0.671 &#xb1; 0.085</td>
<td valign="middle" align="left">0.580 &#xb1; 0.118</td>
<td valign="middle" align="left">0.722 &#xb1; 0.072</td>
<td valign="middle" align="left">0.676 &#xb1; 0.123</td>
<td valign="middle" align="left">
<bold>0.751 &#xb1; 0.129</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">MCC</td>
<td valign="middle" align="left">0.266 &#xb1; 0.131</td>
<td valign="middle" align="left">0.258 &#xb1; 0.114</td>
<td valign="middle" align="left">0.317 &#xb1; 0.113</td>
<td valign="middle" align="left">0.264 &#xb1; 0.135</td>
<td valign="middle" align="left">
<bold>0.367 &#xb1; 0.112</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">AUC</td>
<td valign="middle" align="left">0.669 &#xb1; 0.067</td>
<td valign="middle" align="left">0.663 &#xb1; 0.049</td>
<td valign="middle" align="left">0.703 &#xb1; 0.054</td>
<td valign="middle" align="left">0.678 &#xb1; 0.077</td>
<td valign="middle" align="left">
<bold>0.735 &#xb1; 0.066</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The maximum values of the evaluation metrics among the comparison models are shown in bold. ACC, accuracy; SEN, sensitivity; SPE, specificity; MCC, Matthews correlation coefficient; AUC, area under the receiver operating characteristic curve.</p>
</table-wrap-foot>
</table-wrap>
<p>According to <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, although DeepLION2 had slightly lower performance compared to DeepLION on LUSC samples and TransMIL on skin cutaneous melanoma samples, it generally performed better than the other four models in terms of average AUC validation results across the other eight patient cohorts. The results in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> also demonstrated that DeepLION2 achieved the highest average performance in terms of ACC, SPE, MCC, and AUC among the five models evaluated across the ten cancer types, whereas it obtained the second-highest SEN. As shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, the results highlighted that DeepLION and MINN_SA, which did not consider the correlations among TCRs, exhibited lower SPEs compared to the other models. This suggested that they may be more susceptible to making incorrect predictions on negative samples and having higher false positive rates. On the other hand, TransMIL, which incorporated self-attention to capture TCR correlations, showed higher ACC, SPE, MCC, and AUC, indicating superior classification ability. While BiFormer utilized sparse self-attention to address the sparsity of caTCR distribution, its performance declined compared to TransMIL, probably because of erroneous attention learning brought on by the small sample size. In contrast, DeepLION2 leveraged self-contrastive learning to enhance sparse self-attention learning, resulting in improved predictions of caTCRs and repertoires in terms of all metrics. As a result, the MIL component of DeepLION2 excelled in effectively identifying caTCRs within the repertoire for caTCR prediction by combining sparse self-attention and contrastive learning.</p>
<p>It is noteworthy that the models&#x2019; performance varied among cancer types and that they underperformed in some cases, like LUSC (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). These variations are, in part, due to the TCR feature extraction method. The autoencoder may not have appropriately focused on the motifs when extracting features from the samples in the TCGA dataset in the previous processing (<xref ref-type="bibr" rid="B33">33</xref>). Consequently, poor feature extraction resulted in poor prediction performance. On the other hand, this phenomenon might have been influenced by the heterogeneity among cancer types. Simultaneously, we conjectured that, despite their similar functions, caTCRs in the repertoires of cancer types with low performances differed significantly in sequence form as a result of the structural folding of proteins. The variation in sequences of caTCRs made it difficult for computational methods to predict with accuracy.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>DeepLION2 advances prediction of cancer-associated TCRs and repertoires based on TCR sequencing data</title>
<p>To thoroughly assess the models&#x2019; performance in predicting caTCRs and TCR repertoires using raw TCR-seq data, we conducted experiments on the Geneplus dataset. We employed 5-4-fold nested cross-validation to test all the models on the THCA, LUCA, and GICA patient cohorts from the Geneplus dataset. The AUC results of the models on the three cohorts are shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>, and the validation results of all metrics are shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>The AUC results of models on 3 cancer type samples from Geneplus. THCA, thyroid cancer; LUCA, lung cancer; GICA: gastrointestinal cancer; AUC, area under the receiver operating characteristic curve.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1345586-g003.tif"/>
</fig>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>The validation results of models on 3 cancer type samples from Geneplus.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" colspan="7" align="center">THCA</th>
</tr>
<tr>
<th valign="middle" align="left"/>
<th valign="middle" align="left">DeepTCR</th>
<th valign="middle" align="left">DeepLION</th>
<th valign="middle" align="left">MINN_SA</th>
<th valign="middle" align="left">TransMIL</th>
<th valign="middle" align="left">BiFormer</th>
<th valign="middle" align="left">DeepLION2</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">ACC</td>
<td valign="bottom" align="left">0.733 &#xb1; 0.036</td>
<td valign="bottom" align="left">0.835 &#xb1; 0.039</td>
<td valign="bottom" align="left">0.740 &#xb1; 0.087</td>
<td valign="bottom" align="left">0.816 &#xb1; 0.043</td>
<td valign="bottom" align="left">0.840 &#xb1; 0.024</td>
<td valign="bottom" align="left">
<bold>0.886 &#xb1; 0.035</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">SEN</td>
<td valign="bottom" align="left">0.481 &#xb1; 0.230</td>
<td valign="bottom" align="left">0.722 &#xb1; 0.101</td>
<td valign="bottom" align="left">0.542 &#xb1; 0.306</td>
<td valign="bottom" align="left">0.704 &#xb1; 0.111</td>
<td valign="bottom" align="left">0.729 &#xb1; 0.059</td>
<td valign="bottom" align="left">
<bold>0.751 &#xb1; 0.091</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">SPE</td>
<td valign="bottom" align="left">0.891 &#xb1; 0.170</td>
<td valign="bottom" align="left">0.908 &#xb1; 0.016</td>
<td valign="bottom" align="left">0.874 &#xb1; 0.090</td>
<td valign="bottom" align="left">0.887 &#xb1; 0.052</td>
<td valign="bottom" align="left">0.911 &#xb1; 0.041</td>
<td valign="bottom" align="left">
<bold>0.973 &#xb1; 0.022</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">MCC</td>
<td valign="bottom" align="left">0.457 &#xb1; 0.066</td>
<td valign="bottom" align="left">0.650 &#xb1; 0.087</td>
<td valign="bottom" align="left">0.447 &#xb1; 0.191</td>
<td valign="bottom" align="left">0.612 &#xb1; 0.091</td>
<td valign="bottom" align="left">0.662 &#xb1; 0.050</td>
<td valign="bottom" align="left">
<bold>0.765 &#xb1; 0.075</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">AUC</td>
<td valign="bottom" align="left">0.860 &#xb1; 0.042</td>
<td valign="bottom" align="left">0.892 &#xb1; 0.043</td>
<td valign="bottom" align="left">0.843 &#xb1; 0.053</td>
<td valign="bottom" align="left">0.888 &#xb1; 0.036</td>
<td valign="bottom" align="left">0.917 &#xb1; 0.012</td>
<td valign="bottom" align="left">
<bold>0.933 &#xb1; 0.044</bold>
</td>
</tr>
</tbody>
<tbody>
<tr>
<th valign="middle" colspan="7" align="center">LUCA</th>
</tr>
<tr>
<th valign="middle" align="left"/>
<th valign="middle" align="left">DeepTCR</th>
<th valign="middle" align="left">DeepLION</th>
<th valign="middle" align="left">MINN_SA</th>
<th valign="middle" align="left">TransMIL</th>
<th valign="middle" align="left">BiFormer</th>
<th valign="middle" align="left">DeepLION2</th>
</tr>
</tbody>
<tbody>
<tr>
<td valign="middle" align="left">ACC</td>
<td valign="bottom" align="left">0.721 &#xb1; 0.080</td>
<td valign="bottom" align="left">0.750 &#xb1; 0.072</td>
<td valign="bottom" align="left">0.655 &#xb1; 0.051</td>
<td valign="bottom" align="left">0.757 &#xb1; 0.063</td>
<td valign="bottom" align="left">0.768 &#xb1; 0.050</td>
<td valign="bottom" align="left">
<bold>0.809 &#xb1; 0.050</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">SEN</td>
<td valign="bottom" align="left">0.393 &#xb1; 0.186</td>
<td valign="bottom" align="left">0.653 &#xb1; 0.109</td>
<td valign="bottom" align="left">0.620 &#xb1; 0.341</td>
<td valign="bottom" align="left">0.687 &#xb1; 0.123</td>
<td valign="bottom" align="left">0.716 &#xb1; 0.088</td>
<td valign="bottom" align="left">
<bold>0.736 &#xb1; 0.089</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">SPE</td>
<td valign="bottom" align="left">
<bold>0.968 &#xb1; 0.044</bold>
</td>
<td valign="bottom" align="left">0.810 &#xb1; 0.082</td>
<td valign="bottom" align="left">0.711 &#xb1; 0.208</td>
<td valign="bottom" align="left">0.814 &#xb1; 0.070</td>
<td valign="bottom" align="left">0.812 &#xb1; 0.050</td>
<td valign="bottom" align="left">0.865 &#xb1; 0.037</td>
</tr>
<tr>
<td valign="middle" align="left">MCC</td>
<td valign="bottom" align="left">0.463 &#xb1; 0.102</td>
<td valign="bottom" align="left">0.470 &#xb1; 0.158</td>
<td valign="bottom" align="left">0.358 &#xb1; 0.092</td>
<td valign="bottom" align="left">0.505 &#xb1; 0.114</td>
<td valign="bottom" align="left">0.525 &#xb1; 0.098</td>
<td valign="bottom" align="left">
<bold>0.606 &#xb1; 0.091</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">AUC</td>
<td valign="bottom" align="left">0.836 &#xb1; 0.013</td>
<td valign="bottom" align="left">0.791 &#xb1; 0.075</td>
<td valign="bottom" align="left">0.788 &#xb1; 0.033</td>
<td valign="bottom" align="left">0.820 &#xb1; 0.067</td>
<td valign="bottom" align="left">0.836 &#xb1; 0.039</td>
<td valign="bottom" align="left">
<bold>0.880 &#xb1; 0.030</bold>
</td>
</tr>
</tbody>
<tbody>
<tr>
<th valign="middle" colspan="7" align="center">GICA</th>
</tr>
<tr>
<th valign="middle" align="left"/>
<th valign="middle" align="left">DeepTCR</th>
<th valign="middle" align="left">DeepLION</th>
<th valign="middle" align="left">MINN_SA</th>
<th valign="middle" align="left">TransMIL</th>
<th valign="middle" align="left">BiFormer</th>
<th valign="middle" align="left">DeepLION2</th>
</tr>
</tbody>
<tbody>
<tr>
<td valign="middle" align="left">ACC</td>
<td valign="bottom" align="left">0.657 &#xb1; 0.034</td>
<td valign="bottom" align="left">0.650 &#xb1; 0.021</td>
<td valign="bottom" align="left">0.647 &#xb1; 0.033</td>
<td valign="bottom" align="left">0.681 &#xb1; 0.073</td>
<td valign="bottom" align="left">0.708 &#xb1; 0.032</td>
<td valign="bottom" align="left">
<bold>0.715 &#xb1; 0.061</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">SEN</td>
<td valign="bottom" align="left">0.084 &#xb1; 0.093</td>
<td valign="bottom" align="left">0.292 &#xb1; 0.134</td>
<td valign="bottom" align="left">0.050 &#xb1; 0.076</td>
<td valign="bottom" align="left">0.286 &#xb1; 0.172</td>
<td valign="bottom" align="left">
<bold>0.470 &#xb1; 0.078</bold>
</td>
<td valign="bottom" align="left">0.288 &#xb1; 0.139</td>
</tr>
<tr>
<td valign="middle" align="left">SPE</td>
<td valign="bottom" align="left">0.983 &#xb1; 0.027</td>
<td valign="bottom" align="left">0.865 &#xb1; 0.077</td>
<td valign="bottom" align="left">
<bold>0.993 &#xb1; 0.016</bold>
</td>
<td valign="bottom" align="left">0.922 &#xb1; 0.054</td>
<td valign="bottom" align="left">0.854 &#xb1; 0.092</td>
<td valign="bottom" align="left">0.970 &#xb1; 0.033</td>
</tr>
<tr>
<td valign="middle" align="left">MCC</td>
<td valign="bottom" align="left">0.132 &#xb1; 0.137</td>
<td valign="bottom" align="left">0.193 &#xb1; 0.074</td>
<td valign="bottom" align="left">0.084 &#xb1; 0.141</td>
<td valign="bottom" align="left">0.243 &#xb1; 0.203</td>
<td valign="bottom" align="left">0.362 &#xb1; 0.090</td>
<td valign="bottom" align="left">
<bold>0.379 &#xb1; 0.074</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">AUC</td>
<td valign="bottom" align="left">0.666 &#xb1; 0.085</td>
<td valign="bottom" align="left">0.617 &#xb1; 0.063</td>
<td valign="bottom" align="left">0.654 &#xb1; 0.067</td>
<td valign="bottom" align="left">0.704 &#xb1; 0.053</td>
<td valign="bottom" align="left">0.724 &#xb1; 0.024</td>
<td valign="bottom" align="left">
<bold>0.763 &#xb1; 0.032</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The maximum values of the evaluation metrics among the comparison models are shown in bold. THCA, thyroid cancer; LUCA, lung cancer; GICA: gastrointestinal cancer; ACC, accuracy; SEN, sensitivity; SPE, specificity; MCC, Matthews correlation coefficient; AUC, area under the receiver operating characteristic curve.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>DeepLION2 showed superior performance compared to the other models in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>, with higher average AUC validation results across the three cancer patient cohorts. The results in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> further confirmed DeepLION2&#x2019;s consistent superiority, achieving impressive AUC values of 0.933, 0.880, and 0.763 for the THCA, LUCA, and GICA samples, respectively. Compared to DeepTCR, DeepLION and MINN_SA, TransMIL, BiFormer, and DeepLION2 exhibited better overall prediction performance by considering the correlations among TCRs in the repertoire. BiFormer, which addressed the sparsity of caTCRs and aimed to exclude unrelated TCRs, achieved higher ACCs, SENs, MCCs, and AUCs than TransMIL. However, its SPE performance on LUCA and GICA was weaker. To enhance attention learning, DeepLION2 employed self-contrastive learning during training, resulting in significant improvement in SPE metrics compared to BiFormer without compromising SEN metrics.</p>
<p>In comparison to the prediction performances on the preprocessed samples of the TCGA dataset (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;1</bold>
</xref>), DeepLION2 could produce more accurate predictions on these raw TCR-seq data because of the proper TCR antigen-specificity method. The AUC values of the predictions on three cohorts of the Geneplus dataset were all higher than the average AUC value of the predictions on the TCGA dataset (0.933, 0.880, and 0.763 for THCA, LUCA, and GICA &gt; 0.735 for TCGA). Considering comparisons between samples of the same cancer type, the AUC value of the prediction on the LUCA cohort, consisting of LUAD and LUSC samples, was much higher than those on the LUAD and LUSC samples of the TCGA dataset (0.880 for LUCA &gt; 0.639 and 0.498 for LUAD and LUSC). Although DeepLION2 performed exceptionally well on THCA samples, its performance was comparatively lower on the other two samples. This could be attributed to the inclusion of multiple cancer types or subtypes within the positive samples of LUCA and GICA, as well as the specificity of caTCRs for different cancer types/subtypes. Nevertheless, DeepLION2 consistently demonstrated high SPEs across all three cohorts, indicating its potential for cancer screening. Overall, DeepLION2 showcased a more accurate prediction of caTCRs and repertoires using TCR-seq data from patients with the same cancer type.</p>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>DeepLION2 unveils cancer-associated TCRs with key motifs for antigen-specific recognition in cancer repertoires</title>
<p>DeepLION2 could not only accurately predict the cancer status of patient samples but also identify caTCRs using TCR scores. Additionally, it could pinpoint key motifs of TCRs by calculating motif scores based on the weights of the trained model. In our experiments, we employed the trained models on test samples from THCA patient cohorts to reveal the associated cancer-specific TCRs along with their motifs.</p>
<p>Initially, we selected TCRs with identical CDR3 sequences but from samples with different labels to assess whether considering inter-TCR correlations could enhance the model&#x2019;s performance when encountering these TCRs. The prediction results of DeepLION and TransMIL on these TCRs are shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>. DeepLION, without considering TCR correlations, yielded ambiguous predictions for these TCRs, hovering around 0.5. In contrast, when employing self-attention to account for inter-TCR correlations, TransMIL provided distinct predictions based on their contextual information. It is worth noting that TransMIL predicted low scores for TCRs with the CDR3 sequence &#x201c;CASSSSGTYGYTF&#x201d; from cancer and non-cancer samples, which suggested that within the cancer repertoire, this specific TCR might be considered as a background TCR unrelated to cancer.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>DeepLION2 unveils cancer-associated TCRs with key motifs for antigen-specific recognition in cancer repertoires based on the THCA patient cohorts. <bold>(A)</bold> The DeepLION and TransMIL predictions, the probability that a sequence is cancer-associated, on TCRs with identical sequences but from samples with different labels. Sequences from cancer repertoires are indicated by red, whereas those from non-cancer repertoires are indicated by blue. <bold>(B)</bold> The length distributions of the highest scoring motifs and all positive motifs (motif score &gt; 0.5) in each TCR based on the predictions of DeepLION2. <bold>(C)</bold> DeepLION2 reveals top scoring TCRs (TCR score &gt; 0.999, totally 41 sequences) and visualizes their key motifs. Larger, red-colored amino acids signify the model&#x2019;s prediction of a more substantial role in antigen-binding.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1345586-g004.tif"/>
</fig>
<p>Then, we analyzed the prediction results of DeepLION2 for TCRs with their motifs. We identified TCRs with scores above 0.5 within the cancer cohorts, indicating their potential likelihood of being caTCRs. The results indicated that the predominant length of the highest-scoring motif, most contributing to antigen-specificity within each of these TCRs, is 3 (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>). This finding is consistent with previous methods of preprocessing sequences into 3-length fragments to identify crucial motifs (<xref ref-type="bibr" rid="B10">10</xref>, <xref ref-type="bibr" rid="B12">12</xref>). However, when considering all positive motifs detected by DeepLION2 (motif score &gt; 0.5), their lengths ranged from 2 to 7, aligning with ratios observed in previous X-ray crystal structure analyses (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref>) (<xref ref-type="bibr" rid="B14">14</xref>). Consequently, for TCR antigen-specificity extraction, it is essential to consider motifs of various lengths.</p>
<p>Ultimately, based on the scores of all positive motifs, we computed the amino acid weights of TCRs with a score &gt; 0.999, which were highly probable caTCRs (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4C</bold>
</xref>). This analysis unveiled specific sequence segments DeepLION2 prioritized during predictions. Our assessment of 41 TCRs revealed the model&#x2019;s consistent emphasis on the middle and rear sections of sequences, with less focus on the initial section containing similar amino acids like &#x201c;CA&#x201d; or &#x201c;CS&#x201d;. Moreover, it exhibited limited attention towards the final amino acid, &#x201c;F,&#x201d; except in certain specific combinations. Given our typical expectation of a greater emphasis on the middle sections of sequences due to their higher diversity, it&#x2019;s intriguing that DeepLION2 directed its focus toward the rear sections of specific TCRs. While the diversity of amino acids in the CDR3 tail region is generally lower compared to those in the middle, and the rear sections of different TCRs might display higher similarity, certain scenarios suggest that amino acids in the rear sections could interact with specific parts of the antigenic peptide, potentially serving unique binding functions. On one hand, in many recent studies on TCR-peptide binding prediction, the prediction approaches have more or less reported a focus on amino acids in the rear sections of CDR3 sequences (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B40">40</xref>). On the other hand, some specific motifs in the rear sections were observed to appear more frequently in caTCRs compared to other cancer-unrelated TCRs, implying that we cannot ignore their important role in the cancer-associated antigen-binding process. For instance, the motif &#x201c;NVLT&#x201d;, frequently identified in the rear sections by DeepLION2 (presented in 9 out of 41 TCRs), appeared in 4.9% (145/2969) of caTCRs within the McPAS-TCR database, which is higher than the 2.1% (652/30714) occurrence observed in other TCRs (<xref ref-type="bibr" rid="B24">24</xref>). As a result, it&#x2019;s logical for DeepLION2 to focus on the amino acids in the rear sections of CDR3 sequences, implying their potential significance in recognizing cancer-related antigens.</p>
<p>For further analysis of the identified TCRs and motifs by DeepLION2, we cross-referenced them with the CEDAR and McPAS-TCR databases, renowned for their collection of known caTCRs (<xref ref-type="bibr" rid="B24">24</xref>, <xref ref-type="bibr" rid="B41">41</xref>). We first searched for the 41 TCRs in two databases, but we didn&#x2019;t find identical sequences in either caTCRs or TCRs unrelated to cancer, which may be due to the high diversity of TCRs. Meanwhile, because different types of cancer are highly heterogeneous, it is reasonable that these 41 TCRs specific to THCA were not present in these databases for cancer, containing few TCRs for THCA. Next, upon investigating the motifs that DeepLION2 highlighted in the databases, we observed that certain motifs revealed by DeepLION2 appeared in caTCRs in both two databases. And we also observed that some motifs occurred more frequently in caTCRs compared to other TCRs, such as &#x201c;NVLT&#x201d; as previously mentioned. Some motifs, such as &#x201c;QDPGS&#x201d; and &#x201c;QDPGN&#x201d; (in #10 and 18 TCRs), were even exclusive to caTCRs in the McPAS-TCR database, indicating their potential as THCA-specific biomarkers and promising targets for cancer immunotherapy. Furthermore, the model&#x2019;s preference for non-adjacent amino acids in most TCRs could be attributed to the structural folding of proteins, where amino acids binding to antigen peptides are not sequentially adjacent.</p>
</sec>
<sec id="s3_6">
<label>3.6</label>
<title>Impact of hyperparameters on DeepLION2 prediction performance</title>
<p>Hyperparameters play a crucial role in the performance of a model. We conducted ablation experiments about the important hyperparameters <inline-formula>
<mml:math display="inline" id="im51">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>'</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, <italic>h</italic>, <italic>k<sub>r</sub>
</italic> in DeepLION2 to validate their influence on model performance. In each group of ablation experiments, we changed only the hyperparameters to be observed while keeping the other hyperparameters unchanged and employed 5-4-fold nested cross-validation to test the models on the THCA patient cohorts. The metric AUC was used to evaluate the models and the validation results are shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. According to the results, we observed that the model performance was overall stable and unaffected by these hyperparameter changes. It is worth noting that multi-head self-attention did not achieve a higher accuracy than one-head self-attention in caTCR prediction, which may be due to the small sample size of TCR-seq data. Among the other hyperparameters, <italic>N</italic> was discussed in DeepLION (<xref ref-type="bibr" rid="B17">17</xref>) and set to 100 for the tradeoff between model performance and computational cost. <italic>l<sub>r</sub>
</italic> was usually set to 0.001, whereas due to the use of validation sets and the early stopping approach, <italic>e</italic> would not affect the model performance as long as the model converged during the training process.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>The AUC results of models with different hyperparameters on THCA samples. <bold>(A)</bold> The AUC results of the DeepLION2 models with different ratios of sparse self-attention. <bold>(B)</bold> The AUC results of the DeepLION2 models with different dimensions of self-attention. <bold>(C)</bold> The AUC results of the DeepLION2 models with the dimension of self-attention as 10 and different head numbers of self-attention. <bold>(D)</bold> The AUC results of the DeepLION2 models with the dimension of self-attention as 20 and different head numbers of self-attention.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1345586-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>In this study, we developed a novel deep MIL learning method, named DeepLION2, for improving the prediction of caTCRs and repertoires, which incorporated content-based sparse attention and contrastive learning in its MIL part. Compared to the existing methods, it used sparse self-attention to fully consider the correlations among TCRs and avoided incorrectly predicting TCRs with the same/similar CDR3 sequences as caTCRs. Furthermore, to ensure that the model correctly focused on caTCRs, it used the self-contrastive learning mechanism to improve attention learning. To validate the improvement of DeepLION2, we collected patient samples of more than ten cancer types from TCGA and Geneplus. The results indicated that DeepLION2 generally outperformed the comparison models across the preprocessed ten cancer samples from TCGA. Moreover, the results on the raw TCR-seq data of three cancer patient cohorts from Geneplus also highlighted that DeepLION2 could advance the prediction of caTCRs and repertoires, where its AUC values reached notably 0.933, 0.880, and 0.763 on the THCA, LUCA and GICA patient cohorts, respectively.</p>
<p>To mitigate overfitting concerns, we took several steps in our experiments. Firstly, we simplified the model structure by using only one-layer self-attention/sparse self-attention, which helps prevent overfitting when the training data is limited. Additionally, we incorporated random dropout with a rate of 40% during training, a well-established technique known for effectively reducing overfitting and widely used in various machine learning models (<xref ref-type="bibr" rid="B42">42</xref>). Furthermore, we employed the early-stopping approach to prevent the model from overtraining. By monitoring the model&#x2019;s performance on validation sets, we stopped the training process at an appropriate time to avoid performance degradation on the test sets (<xref ref-type="bibr" rid="B43">43</xref>). This approach helps ensure that the model generalizes well to unseen data. Moreover, the utilization of nested cross-validation, a robust and unbiased validation technique, further reinforced the outstanding performance of DeepLION2 in predicting caTCRs. By validating the model on multiple folds of the data, we obtained reliable and comprehensive performance estimates, enhancing the confidence in the model&#x2019;s predictive capabilities.</p>
<p>In the comparison experiments conducted on both the TCGA and Geneplus datasets, DeepLION2 consistently outperformed existing methods. This can be attributed to its utilization of content-based sparse self-attention to effectively model the correlations among TCRs, along with the incorporation of self-contrastive learning to enhance attention learning. Notably, as described in Section 3.4, the performance of the models on the preprocessed samples from the TCGA dataset was inferior to that on the raw TCR-seq samples from the Geneplus dataset. This discrepancy can be attributed to the differences in the approaches used for TCR antigen-specificity extraction between the two datasets. In the TCGA dataset, stacked auto-encoders were employed for TCR feature extraction. However, this approach did not take into account the key motifs of different lengths present in the TCR CDR3 sequences. On the other hand, the raw TCR-seq samples from the Geneplus dataset were processed using a convolutional network with filters of different sizes, allowing for the handling of fragments with varying lengths in TCRs. Hence, the methodology used for TCR antigen-specificity extraction plays a crucial role in predicting caTCRs, and it is this aspect that contributes to the outstanding performance of DeepLION2.</p>
<p>While proficiently discerning cancer-associated patient repertoires, DeepLION2 concurrently identifies caTCRs within these repertoires, shedding light on key motifs. The model&#x2019;s emphasis on the rear sections of CDR3 sequences from the 41 TCRs in THCA patient cohorts aligns with previous research more or less focusing on the amino acids in such sections. Notably, certain motifs occurring more frequently in caTCRs compared to non-cancer-related TCRs underscore the significance of DeepLION2&#x2019;s attention to these rear-section amino acids. It is crucial not to overlook these amino acids when studying TCR-peptide binding. It&#x2019;s worth noting that DeepLION2&#x2019;s focus on the 41 TCRs and their motifs does not necessarily imply their direct association with cancer or involvement in binding to cancerous antigens. The attention mechanism indicates the features contributing to the classification between cancerous and non-cancerous repertoires, suggesting potential caTCRs and amino acids relevant to cancer antigen recognition and binding. For a deeper analysis, we cross-referenced these results with existing cancer databases. Due to the vast diversity of TCRs and the heterogeneity of cancers, the 41 TCRs from THCA did not appear in the caTCR or non-cancer-related TCR lists in the databases. Nevertheless, certain motifs identified by DeepLION2 were found in caTCRs in both databases. Additionally, some motifs were more prevalent in caTCRs, with a few exclusive to caTCRs. These findings hint at the potential of these motifs as THCA-specific biomarkers, supporting the validity of slicing TCRs into motifs for consideration.</p>
<p>In future work, we aim to further validate the performance of DeepLION2 by applying it to a broader range of cancer types. We acknowledge that the performance of DeepLION2 experienced a decline when samples contained multiple cancer types and when the size of the training samples was smaller. To address this, we plan to enhance the model to more effectively extract the specificity of caTCRs from limited data, thereby improving its performance in such scenarios. Furthermore, we recognize that the presence of noise in TCR-seq data poses a limitation on the model&#x2019;s performance. This is an important issue that we intend to address in future research. By developing techniques to mitigate the impact of noise in TCR-seq data, we aim to enhance the robustness and accuracy of DeepLION2 for predicting caTCRs and advancing its practical utility in clinical settings. In addition, it has been recognized that the &#x3b1; chain, a constituent of the TCR along with the &#x3b2; chain, also plays a significant role in the recognition of antigens. For a more comprehensive understanding of the antigen recognition mechanism of the TCR, we will further consider the &#x3b1; chain and develop models to support the analysis of both chains.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusion</title>
<p>DeepLION2 is a groundbreaking deep MIL framework that integrates content-based sparse attention and contrastive learning to capture TCR correlations in a repertoire. It outperforms existing methods in accurate caTCR and repertoire prediction from TCR-seq data. Additionally, it can unveil potential caTCRs and their crucial motifs. DeepLION2 enables effective repertoire classification, potentially supporting cancer detection and facilitating personalized cancer immunotherapy.</p>
</sec>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>DeepLION2 is available on GitHub, at <uri xlink:href="https://github.com/Bioinformatics7181/DeepLION2">https://github.com/Bioinformatics7181/DeepLION2</uri>, for academic use only. The preprocessed samples of the TCGA dataset were from Xiong&#x2019;s study (<xref ref-type="bibr" rid="B33">33</xref>), which can be found at <uri xlink:href="https://github.com/danyixiong/MIL_Comparative_Study">https://github.com/danyixiong/MIL_Comparative_Study</uri>. In the context of the Geneplus dataset, the THCA TCR-seq samples were from Lan&#x2019;s study (<xref ref-type="bibr" rid="B34">34</xref>), which can be found in NCBI, at <uri xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA642967">https://www.ncbi.nlm.nih.gov/bioproject/PRJNA642967</uri>, whereas the LUCA samples were from Li&#x2019;s study (<xref ref-type="bibr" rid="B36">36</xref>) and the GICA samples were from Ji&#x2019;s study (<xref ref-type="bibr" rid="B35">35</xref>). All the processed data used in the experiments can be found at <uri xlink:href="https://github.com/Bioinformatics7181/DeepLION2">https://github.com/Bioinformatics7181/DeepLION2</uri>.</p>
</sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>Ethical approval was not required for the studies on humans in accordance with the local legislation and institutional requirements because only commercially available established cell lines were used.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>XQ: Data curation, Methodology, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. GY: Data curation, Investigation, Methodology, Validation, Writing &#x2013; review &amp; editing. FL: Writing &#x2013; review &amp; editing. XuZ: Supervision, Writing &#x2013; review &amp; editing. XiZ: Writing &#x2013; review &amp; editing. XL: Writing &#x2013; review &amp; editing. XX: Writing &#x2013; review &amp; editing. TW: Supervision, Writing &#x2013; review &amp; editing. JW: Funding acquisition, Supervision, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This research was funded by the Natural Science Basic Research Program of Shaanxi, grant number 2020JC-01.</p>
</sec>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Author XX was employed by the company Geneplus-Shenzhen.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fimmu.2024.1345586/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fimmu.2024.1345586/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table_1.xlsx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"/>
</sec>
<fn-group>
<title>Abbreviations</title>
<fn fn-type="abbr">
<p>TCR, T cell receptor; caTCR, cancer-associated T cell receptor; AIRR-seq, The Adaptive Immune Receptor Repertoire sequencing; MIL, multi-instance learning; CDR3, complementarity determining region 3; TCR-seq, T cell receptor sequencing; ROC, receiver operating characteristic; AUC, area under the receiver operating characteristic curve; THCA, thyroid cancer; LUCA, lung cancer; GICA, gastrointestinal cancer; TCGA, The Cancer Genome Atlas; Geneplus, clinical database of Geneplus Technology Ltd. in Shenzhen; LUSC, lung squamous cell carcinoma; LUAD, lung adenocarcinoma; ACC, accuracy; SEN, sensitivity; SPE, specificity; MCC, Matthews correlation coefficient.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gubin</surname> <given-names>MM</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Schuster</surname> <given-names>H</given-names>
</name>
<name>
<surname>Caron</surname> <given-names>E</given-names>
</name>
<name>
<surname>Ward</surname> <given-names>JP</given-names>
</name>
<name>
<surname>Noguchi</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>Checkpoint blockade cancer immunotherapy targets tumor-specific mutant antigens</article-title>. <source>Nature</source>. (<year>2014</year>) <volume>515</volume>:<page-range>577&#x2013;81</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature13988</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tran</surname> <given-names>E</given-names>
</name>
<name>
<surname>Turcotte</surname> <given-names>S</given-names>
</name>
<name>
<surname>Gros</surname> <given-names>A</given-names>
</name>
<name>
<surname>Robbins</surname> <given-names>PF</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>YC</given-names>
</name>
<name>
<surname>Dudley</surname> <given-names>ME</given-names>
</name>
<etal/>
</person-group>. <article-title>Cancer immunotherapy based on mutation-specific CD4+ T cells in a patient with epithelial cancer</article-title>. <source>Science</source>. (<year>2014</year>) <volume>344</volume>:<page-range>641&#x2013;5</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.1251102</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tumeh</surname> <given-names>PC</given-names>
</name>
<name>
<surname>Harview</surname> <given-names>CL</given-names>
</name>
<name>
<surname>Yearley</surname> <given-names>JH</given-names>
</name>
<name>
<surname>Shintaku</surname> <given-names>IP</given-names>
</name>
<name>
<surname>Taylor</surname> <given-names>EJ</given-names>
</name>
<name>
<surname>Robert</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>PD-1 blockade induces responses by inhibiting adaptive immune resistance</article-title>. <source>Nature</source>. (<year>2014</year>) <volume>515</volume>:<page-range>568&#x2013;71</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature13954</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schreiber</surname> <given-names>RD</given-names>
</name>
<name>
<surname>Old</surname> <given-names>LJ</given-names>
</name>
<name>
<surname>Smyth</surname> <given-names>MJ</given-names>
</name>
</person-group>. <article-title>Cancer immunoediting: integrating immunity&#x2019;s roles in cancer suppression and promotion</article-title>. <source>Science</source>. (<year>2011</year>) <volume>331</volume>:<page-range>1565&#x2013;70</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.1203486</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kvistborg</surname> <given-names>P</given-names>
</name>
<name>
<surname>van Buuren</surname> <given-names>MM</given-names>
</name>
<name>
<surname>Schumacher</surname> <given-names>TN</given-names>
</name>
</person-group>. <article-title>Human cancer regression antigens</article-title>. <source>Curr Opin Immunol</source>. (<year>2013</year>) <volume>25</volume>:<page-range>284&#x2013;90</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.coi.2013.03.005</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chowell</surname> <given-names>D</given-names>
</name>
<name>
<surname>Krishna</surname> <given-names>S</given-names>
</name>
<name>
<surname>Becker</surname> <given-names>PD</given-names>
</name>
<name>
<surname>Cocita</surname> <given-names>C</given-names>
</name>
<name>
<surname>Shu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>X</given-names>
</name>
<etal/>
</person-group>. <article-title>TCR contact residue hydrophobicity is a hallmark of immunogenic CD8+ T cellEpitopes</article-title>. <source>Proc Natl Acad Sci USA</source>. (<year>2015</year>) <volume>112</volume>:<page-range>E1754&#x2013;62</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1500973112</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dhodapkar</surname> <given-names>K</given-names>
</name>
<name>
<surname>Dhodapkar</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Harnessing shared antigens and T-cell receptors in cancer: opportunities and challenges</article-title>. <source>Proc Natl Acad Sci USA</source>. (<year>2016</year>) <volume>113</volume>:<page-range>7944&#x2013;5</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1608860113</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>B</given-names>
</name>
<name>
<surname>Li</surname> <given-names>T</given-names>
</name>
<name>
<surname>Pignon</surname> <given-names>JC</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>B</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Shukla</surname> <given-names>SA</given-names>
</name>
<etal/>
</person-group>. <article-title>Landscape of tumor-infiltrating T cell repertoire of human cancers</article-title>. <source>Nat Genet</source>. (<year>2016</year>) <volume>48</volume>:<page-range>725&#x2013;32</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ng.3581</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kirsch</surname> <given-names>I</given-names>
</name>
<name>
<surname>Vignali</surname> <given-names>M</given-names>
</name>
<name>
<surname>Robins</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>T-cell receptor profiling in cancer</article-title>. <source>Mol Oncol</source>. (<year>2015</year>) <volume>9</volume>:<page-range>2063&#x2013;70</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.molonc.2015.09.003</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cinelli</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Best</surname> <given-names>K</given-names>
</name>
<name>
<surname>Heather</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Reich-Zeliger</surname> <given-names>S</given-names>
</name>
<name>
<surname>Shifrut</surname> <given-names>E</given-names>
</name>
<etal/>
</person-group>. <article-title>Feature selection using a one dimensional na&#xef;ve bayes&#x2019; Classifier increases the accuracy of support vector machine classification of CDR3 repertoires</article-title>. <source>Bioinformatics</source>. (<year>2017</year>) <volume>33</volume>:<page-range>btw771&#x2013;955</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btw771</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Emerson</surname> <given-names>RO</given-names>
</name>
<name>
<surname>DeWitt</surname> <given-names>WS</given-names>
</name>
<name>
<surname>Vignali</surname> <given-names>M</given-names>
</name>
<name>
<surname>Gravley</surname> <given-names>J</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>JK</given-names>
</name>
<name>
<surname>Osborne</surname> <given-names>EJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Immunosequencing identifies signatures of cytomegalovirus exposure history and HLA-mediated effects on the T cell repertoire</article-title>. <source>Nat Genet</source>. (<year>2017</year>) <volume>49</volume>:<page-range>659&#x2013;65</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ng.3822</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Best</surname> <given-names>K</given-names>
</name>
<name>
<surname>Cinelli</surname> <given-names>M</given-names>
</name>
<name>
<surname>Heather</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Reich-Zeliger</surname> <given-names>S</given-names>
</name>
<name>
<surname>Shifrut</surname> <given-names>E</given-names>
</name>
<etal/>
</person-group>. <article-title>Specificity, privacy, and degeneracy in the CD4 T cell receptor repertoire following immunization</article-title>. <source>Front Immunol</source>. (<year>2017</year>) <volume>8</volume>:<elocation-id>430</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2017.00430</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yokota</surname> <given-names>R</given-names>
</name>
<name>
<surname>Kaminaga</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Kobayashi</surname> <given-names>TJ</given-names>
</name>
</person-group>. <article-title>Quantification of inter-sample differences in T-cell receptor repertoires using sequence-based information</article-title>. <source>Front Immunol</source>. (<year>2017</year>) <volume>8</volume>:<elocation-id>1500</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2017.01500</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ostmeyer</surname> <given-names>J</given-names>
</name>
<name>
<surname>Christley</surname> <given-names>S</given-names>
</name>
<name>
<surname>Toby</surname> <given-names>IT</given-names>
</name>
<name>
<surname>Cowell</surname> <given-names>LG</given-names>
</name>
</person-group>. <article-title>Biophysicochemical motifs in T-cell receptor sequences distinguish repertoires from tumor-infiltrating lymphocyte and adjacent healthy tissue</article-title>. <source>Cancer Res</source>. (<year>2019</year>) <volume>79</volume>:<page-range>1671&#x2013;80</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/0008-5472.CAN-18-2292</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beshnova</surname> <given-names>D</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>J</given-names>
</name>
<name>
<surname>Onabolu</surname> <given-names>O</given-names>
</name>
<name>
<surname>Moon</surname> <given-names>B</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>W</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>YX</given-names>
</name>
<etal/>
</person-group>. <article-title>
<italic>De novo</italic> prediction of cancer-associated T cell receptors for noninvasive cancer detection</article-title>. <source>Sci Transl Med</source>. (<year>2020</year>) <volume>12</volume>:<elocation-id>eaaz3738</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/scitranslmed.aaz3738</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sidhom</surname> <given-names>JW</given-names>
</name>
<name>
<surname>Larman</surname> <given-names>HB</given-names>
</name>
<name>
<surname>Pardoll</surname> <given-names>DM</given-names>
</name>
<name>
<surname>Baras</surname> <given-names>AS</given-names>
</name>
</person-group>. <article-title>DeepTCR is a deep learning framework for revealing sequence concepts within T-cell repertoires</article-title>. <source>Nat Commun</source>. (<year>2021</year>) <volume>12</volume>:<fpage>1605</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-021-21879-w</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Qian</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>X</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>DeepLION: deep multi-instance learning improves the prediction of cancer-associated T cell receptors for accurate cancer detection</article-title>. <source>Front Genet</source>. (<year>2022</year>a) <volume>13</volume>:<elocation-id>860510</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fgene.2022.860510</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>T</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>D</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Park</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Multiple instance neural networks based on sparse attention for cancer detection using T-cell receptor sequences</article-title>. <source>BMC Bioinf</source>. (<year>2022</year>) <volume>23</volume>:<fpage>469</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12859-022-05012-2</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sidhom</surname> <given-names>JW</given-names>
</name>
<name>
<surname>Oliveira</surname> <given-names>G</given-names>
</name>
<name>
<surname>Ross-MacDonald</surname> <given-names>P</given-names>
</name>
<name>
<surname>Wind-Rotolo</surname> <given-names>M</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>CJ</given-names>
</name>
<name>
<surname>Pardoll</surname> <given-names>DM</given-names>
</name>
<etal/>
</person-group>. <article-title>Deep learning reveals predictive sequence concepts within immune repertoires to immunotherapy</article-title>. <source>Sci Adv</source>. (<year>2022</year>) <volume>8</volume>:<elocation-id>eabq5089</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/sciadv.abq5089</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>X</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>X</given-names>
</name>
<etal/>
</person-group>. <article-title>TMBcat: A multi-endpoint P-value criterion on different discrepancy metrics for superiorly inferring tumor mutation burden thresholds</article-title>. <source>Front Immunol</source>. (<year>2022</year>) <volume>13</volume>:<elocation-id>995180</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2022.995180</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>JT</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>X</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>ZY</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>B</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>PX</given-names>
</name>
<etal/>
</person-group>. <article-title>Dynamic circulating tumor DNA during chemoradiotherapy predicts clinical outcomes for locally advanced non-small cell lung cancer patients</article-title>. <source>Cancer Cell</source>. (<year>2023</year>) <volume>41</volume>:<page-range>1763&#x2013;73</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ccell.2023.09.007</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>X</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>TMBserval: A statistical explainable learning model reveals weighted tumor mutation burden better categorizing therapeutic benefits</article-title>. <source>Front Immunol</source>. (<year>2023</year>) <volume>14</volume>:<elocation-id>1151755</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2023.1151755</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dietterich</surname> <given-names>TG</given-names>
</name>
<name>
<surname>Lathrop</surname> <given-names>RH</given-names>
</name>
<name>
<surname>Lozano-P&#xe9;rez</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>Solving the multiple instance problem with axis-parallel rectangles</article-title>. <source>Artif Intelligence</source>. (<year>1997</year>) <volume>89</volume>:<fpage>31</fpage>&#x2013;<lpage>71</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/s0004-3702(96)00034-3</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tickotsky</surname> <given-names>N</given-names>
</name>
<name>
<surname>Sagiv</surname> <given-names>T</given-names>
</name>
<name>
<surname>Prilusky</surname> <given-names>J</given-names>
</name>
<name>
<surname>Shifrut</surname> <given-names>E</given-names>
</name>
<name>
<surname>Friedman</surname> <given-names>N</given-names>
</name>
</person-group>. <article-title>McPAS-TCR: a manually curated catalogue of pathology-associated T cell receptor sequences</article-title>. <source>Bioinformatics</source>. (<year>2017</year>) <volume>33</volume>:<page-range>2924&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btx286</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname> <given-names>A</given-names>
</name>
<name>
<surname>Shazeer</surname> <given-names>N</given-names>
</name>
<name>
<surname>Parmar</surname> <given-names>N</given-names>
</name>
<name>
<surname>Uszkoreit</surname> <given-names>J</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>L</given-names>
</name>
<name>
<surname>Gomez</surname> <given-names>AN</given-names>
</name>
<etal/>
</person-group>. <article-title>Attention is all you need</article-title>. <source>Adv Neural Inf Process Syst</source>. (<year>2017</year>) <volume>30</volume>.</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shao</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Bian</surname> <given-names>H</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>Transmil: transformer based correlated multiple instance learning for whole slide image classification</article-title>. <source>Adv Neural Inf Process Syst</source>. (<year>2021</year>) <volume>34</volume>:<page-range>2136&#x2013;47</page-range>.</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scheper</surname> <given-names>W</given-names>
</name>
<name>
<surname>Kelderman</surname> <given-names>S</given-names>
</name>
<name>
<surname>Fanchi</surname> <given-names>LF</given-names>
</name>
<name>
<surname>Linnemann</surname> <given-names>C</given-names>
</name>
<name>
<surname>Bendle</surname> <given-names>G</given-names>
</name>
<name>
<surname>de Rooij</surname> <given-names>MA</given-names>
</name>
<etal/>
</person-group>. <article-title>Low and variable tumor reactivity of the intratumoral TCR repertoire in human cancers</article-title>. <source>Nat Med</source>. (<year>2019</year>) <volume>25</volume>:<fpage>89</fpage>&#x2013;<lpage>94</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41591-018-0266-5</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kidera</surname> <given-names>A</given-names>
</name>
<name>
<surname>Konishi</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Oka</surname> <given-names>M</given-names>
</name>
<name>
<surname>Ooi</surname> <given-names>T</given-names>
</name>
<name>
<surname>Scheraga</surname> <given-names>HA</given-names>
</name>
</person-group>. <article-title>Statistical analysis of the physical properties of the 20 naturally occurring amino acids</article-title>. <source>J Protein Chem</source>. (<year>1985</year>) <volume>4</volume>:<fpage>23</fpage>&#x2013;<lpage>55</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/BF01025492</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Atchley</surname> <given-names>WR</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>J</given-names>
</name>
<name>
<surname>Fernandes</surname> <given-names>AD</given-names>
</name>
<name>
<surname>Dr&#xfc;ke</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>Solving the protein sequence metric problem</article-title>. <source>Proc Natl Acad Sci</source>. (<year>2005</year>) <volume>102</volume>:<page-range>6395&#x2013;400</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.0408677102</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>T</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>A survey of transformers</article-title>. <source>AI Open</source>. (<year>2022</year>) <volume>3</volume>:<page-range>111&#x2013;32</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.aiopen.2022.10.001</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Ke</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Lau</surname> <given-names>RW</given-names>
</name>
</person-group>. <article-title>BiFormer: vision transformer with bi-level routing attention</article-title>. <source>In Proc IEEE/CVF Conf Comput Vision Pattern Recognition</source>. (<year>2023</year>), <page-range>10323&#x2013;33</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR52729.2023.00995</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tomczak</surname> <given-names>K</given-names>
</name>
<name>
<surname>Czerwi&#x144;ska</surname> <given-names>P</given-names>
</name>
<name>
<surname>Wiznerowicz</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Review the cancerGenome atlas (TCGA): an immeasurable source of knowledge</article-title>. <source>Wspolczesna Onkol</source>. (<year>2015</year>) <volume>1A</volume>:<fpage>68</fpage>&#x2013;<lpage>77</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5114/wo.2014.47136</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiong</surname> <given-names>D</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>T</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>A comparative study of multiple instance learning methods for cancer detection using T-cell receptor sequences</article-title>. <source>Comput Struct Biotechnol J</source>. (<year>2021</year>) <volume>19</volume>:<page-range>3255&#x2013;68</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.csbj.2021.05.038</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lan</surname> <given-names>X</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>K</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Ge</surname> <given-names>X</given-names>
</name>
<etal/>
</person-group>. <article-title>TCR-seq identifies distinct repertoires of distant-metastatic and nondistant-metastatic thyroid tumors</article-title>. <source>J Clin Endocrinol Metab</source>. (<year>2020</year>) <volume>105</volume>:<page-range>3036&#x2013;45</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1210/clinem/dgaa452</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname> <given-names>S</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>L</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>C</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>R</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>Z</given-names>
</name>
<etal/>
</person-group>. <article-title>Peripheral blood T-cell receptor repertoire as a predictor of clinical outcomes in gastrointestinal cancer patients treated with PD-1 inhibitor</article-title>. <source>Clin Transl Oncol</source>. (<year>2021</year>) <volume>23</volume>:<page-range>1646&#x2013;56</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12094-021-02562-4</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>M</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Deng</surname> <given-names>S</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S</given-names>
</name>
<name>
<surname>Bai</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Lung cancer-associated T cell repertoire as potential biomarker for early detection of stage I lung cancer</article-title>. <source>Lung Cancer</source>. (<year>2021</year>) <volume>162</volume>:<fpage>16</fpage>&#x2013;<lpage>22</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.lungcan.2021.09.017</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>HY</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>CH</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>TY</given-names>
</name>
<name>
<surname>Horng</surname> <given-names>JT</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>TP</given-names>
</name>
<name>
<surname>Tseng</surname> <given-names>YJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Rapid detection of heterogeneous vancomycin-intermediate staphylococcus aureus based on matrix-assisted laser desorption ionizationTime-of-flight: using a machine learning approach and unbiased validation</article-title>. <source>Front Microbiol</source>. (<year>2018</year>) <volume>9</volume>:<elocation-id>2393</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmicb.2018.02393</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vabalas</surname> <given-names>A</given-names>
</name>
<name>
<surname>Gowen</surname> <given-names>E</given-names>
</name>
<name>
<surname>Poliakoff</surname> <given-names>E</given-names>
</name>
<name>
<surname>Casson</surname> <given-names>AJ</given-names>
</name>
</person-group>. <article-title>Machine learning algorithm validation with a limited sample size</article-title>. <source>PloS One</source>. (<year>2019</year>) <volume>14</volume>:<elocation-id>e0224365</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0224365</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Qian</surname> <given-names>X</given-names>
</name>
<name>
<surname>Tong</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Li</surname> <given-names>F</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>K</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<etal/>
</person-group>. <article-title>AttnTAP: A dual-input framework incorporating the attention mechanism for accurately predicting TCR-peptide binding</article-title>. <source>Front Genet</source>. (<year>2022</year>b) <volume>13</volume>:<elocation-id>942491</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fgene.2022.942491</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>He</surname> <given-names>B</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>F</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Su</surname> <given-names>X</given-names>
</name>
<etal/>
</person-group>. <article-title>DeepAIR: A deep learning framework for effective integration of sequence and 3D structure to enable adaptive immune receptor analysis</article-title>. <source>Sci Adv</source>. (<year>2023</year>) <volume>9</volume>:<elocation-id>eabo5128</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/sciadv.abo5128</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ko&#x15f;alo&#x11f;lu-Yal&#xe7;&#x131;n</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Blazeska</surname> <given-names>N</given-names>
</name>
<name>
<surname>Vita</surname> <given-names>R</given-names>
</name>
<name>
<surname>Carter</surname> <given-names>H</given-names>
</name>
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Schoenberger</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>The cancer epitope database and analysis resource (CEDAR)</article-title>. <source>Nucleic Acids Res</source>. (<year>2023</year>) <volume>51</volume>:<page-range>D845&#x2013;52</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkac902</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srivastava</surname> <given-names>N</given-names>
</name>
<name>
<surname>Hinton</surname> <given-names>G</given-names>
</name>
<name>
<surname>Krizhevsky</surname> <given-names>A</given-names>
</name>
<name>
<surname>Sutskever</surname> <given-names>I</given-names>
</name>
<name>
<surname>Salakhutdinov</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Dropout: a simple way to prevent neural networks from overfitting</article-title>. <source>J Mach Learn Res</source>. (<year>2014</year>) <volume>15</volume>:<page-range>1929&#x2013;58</page-range>.</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Rosasco</surname> <given-names>L</given-names>
</name>
<name>
<surname>Caponnetto</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>On early stopping in gradient descent learning</article-title>. <source>Constr Approx</source>. (<year>2007</year>) <volume>26</volume>:<fpage>289</fpage>&#x2013;<lpage>315</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00365-006-0663-2</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>