<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioinform.</journal-id>
<journal-title>Frontiers in Bioinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioinform.</abbrev-journal-title>
<issn pub-type="epub">2673-7647</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1207380</article-id>
<article-id pub-id-type="doi">10.3389/fbinf.2023.1207380</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioinformatics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Improved prediction of MHC-peptide binding using protein language models</article-title>
<alt-title alt-title-type="left-running-head">Hashemi et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbinf.2023.1207380">10.3389/fbinf.2023.1207380</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Hashemi</surname>
<given-names>Nasser</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2096489/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hao</surname>
<given-names>Boran</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2334313/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ignatov</surname>
<given-names>Mikhail</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Paschalidis</surname>
<given-names>Ioannis Ch.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1381909/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Vakili</surname>
<given-names>Pirooz</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Vajda</surname>
<given-names>Sandor</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2332838/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kozakov</surname>
<given-names>Dima</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2284747/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Division of Systems Engineering</institution>, <institution>Boston University</institution>, <addr-line>Boston</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Electrical and Computer Engineering</institution>, <institution>Boston University</institution>, <addr-line>Boston</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Applied Mathematics and Statistics</institution>, <institution>Stony Brook University</institution>, <addr-line>Stony Brook</addr-line>, <addr-line>NY</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Laufer Center for Physical and Quantitative Biology</institution>, <institution>Stony Brook University</institution>, <addr-line>Stony Brook</addr-line>, <addr-line>NY</addr-line>, <country>United States</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Department of Biomedical Engineering</institution>, <institution>Boston University</institution>, <addr-line>Boston</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Chemistry</institution>, <institution>Boston University</institution>, <addr-line>Boston</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/163898/overview">Igor N. Berezovsky</ext-link>, Bioinformatics Institute (A&#x2217;STAR), Singapore</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/639086/overview">Jian Zhang</ext-link>, Xinyang Normal University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/402813/overview">Giuseppe Maccari</ext-link>, The Pirbright Institute, United Kingdom</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Nasser Hashemi, <email>nhashemi@bu.edu</email>; Dima Kozakov, <email>midas@laufercenter.org</email>
</corresp>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>08</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>3</volume>
<elocation-id>1207380</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>04</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>08</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Hashemi, Hao, Ignatov, Paschalidis, Vakili, Vajda and Kozakov.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Hashemi, Hao, Ignatov, Paschalidis, Vakili, Vajda and Kozakov</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Major histocompatibility complex Class I (MHC-I) molecules bind to peptides derived from intracellular antigens and present them on the surface of cells, allowing the immune system (T cells) to detect them. Elucidating the process of this presentation is essential for regulation and potential manipulation of the cellular immune system. Predicting whether a given peptide binds to an MHC molecule is an important step in the above process and has motivated the introduction of many computational approaches to address this problem. NetMHCPan, a pan-specific model for predicting binding of peptides to any MHC molecule, is one of the most widely used methods which focuses on solving this binary classification problem using shallow neural networks. The recent successful results of Deep Learning (DL) methods, especially Natural Language Processing (NLP-based) pretrained models in various applications, including protein structure determination, motivated us to explore their use in this problem. Specifically, we consider the application of deep learning models pretrained on large datasets of protein sequences to predict MHC Class I-peptide binding. Using the standard performance metrics in this area, and the same training and test sets, we show that our models outperform NetMHCpan4.1, currently considered as the-state-of-the-art.</p>
</abstract>
<kwd-group>
<kwd>MHC class I</kwd>
<kwd>deep learning</kwd>
<kwd>transformers</kwd>
<kwd>natural language processing</kwd>
<kwd>cellular immune system</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Protein Bioinformatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Major Histocompatibility Complex molecules (MHC) are large cell surface proteins that play a key role in immune response by detecting and responding to foreign proteins and antigens. An MHC molecule detects and binds to a peptide (a small fragment of a protein derived from an antigen), creating a peptide-MHC complex, and presents it to the surface of the cell; then, based on the interactions between this complex and the T cell receptor at the cell surface, an immune response is triggered to control the compromised cell (<xref ref-type="bibr" rid="B31">Maimela et al., 2019</xref>; <xref ref-type="bibr" rid="B24">Janeway et al., 2001</xref>; <xref ref-type="bibr" rid="B44">Teraguchi et al., 2020</xref>; <xref ref-type="bibr" rid="B36">Ong et al., 2021</xref>). MHC molecules are classified into two classes: (i) MHC Class I which controls non-self intracellular antigens by presenting antigenic peptides (of 8&#x2013;14 sequence length) to cytotoxic T cell lymphocytes (CD8<sup>&#x2b;</sup> TCR) and (ii) MHC Class II, which controls extracellular antigens by presenting antigenic peptides (of 13&#x2013;25 sequence length) to helper T cell lymphocytes (CD4<sup>&#x2b;</sup> TCR). One of the main steps in studying the role of the MHC molecules in the immune system is developing insights into the interactions of the MHC molecules and non-self pathogen peptides, referred to as MHC-peptide binding (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>). MHC-peptide binding prediction plays an important role in vaccine design and studies of infectious diseases, autoimmunity, and cancer therapy (<xref ref-type="bibr" rid="B35">O&#x2019;Donnell et al., 2020</xref>; <xref ref-type="bibr" rid="B20">Grebenkin et al., 2020</xref>).</p>
<p>There are two basic experimental methods to study MHC-peptide binding: (i) Peptide-MHC binding affinity (BA) assays in which, given a peptide, binding preferences of different MHC molecules to the peptide are measured (<xref ref-type="bibr" rid="B46">Townsend et al., 1990</xref>); (ii) MHC associated eluted ligands (EL) generated by Liquid Chromatography Mass Spectrometry (LC-MS) in which, based on a single experiment, a large number of eluted ligands corresponding to an MHC are identified (<xref ref-type="bibr" rid="B10">Caron et al., 2015</xref>). Compared to the BA method, the EL method is highly accurate and thorough and it is a reliable way to determine the peptides included in the immunopeptidome (namely, the entire set of peptides forming MHC-peptides complexes (<xref ref-type="bibr" rid="B3">Alvarez et al., 2019</xref>)). Both methods, however, are labor-intensive and time-consuming. As a result, a number of computational methods have been developed to predict MHC-peptide binding (<xref ref-type="bibr" rid="B7">Boehm et al., 2019</xref>). These methods include heuristic approaches using MHC allele&#x2013;specific motifs to identify potential ligands in a protein sequence (<xref ref-type="bibr" rid="B9">Bui et al., 2005</xref>), supervised machine learning approaches, including artificial neural networks (ANN) (<xref ref-type="bibr" rid="B34">Nielsen et al., 2003</xref>), hidden Markov models (HMM) (<xref ref-type="bibr" rid="B52">Zhang et al., 2006</xref>), and regression models (<xref ref-type="bibr" rid="B37">Parker et al., 1994</xref>; <xref ref-type="bibr" rid="B14">Doytchinova and Flower, 2001</xref>). The performance of these machine learning methods increases with the amount of data available in epitope databases such as SysteMHC (<xref ref-type="bibr" rid="B43">Shao et al., 2018</xref>) and Immune Epitope Database (IEDB) (<xref ref-type="bibr" rid="B49">Vita et al., 2019</xref>). While some of these methods are trained for only one specific MHC allele (known as allele-specific methods), there are more generalized models (pan-specific methods) where a single model covers all of alleles of interest. The methods are also categorized by the type of predicted variables. Among these methods, some have been shown to be more promising, such as NetMHCpan (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>), DeepLigand (<xref ref-type="bibr" rid="B51">Zeng and Gifford, 2019</xref>), and MHCflurry (<xref ref-type="bibr" rid="B35">O&#x2019;Donnell et al., 2020</xref>; <xref ref-type="bibr" rid="B4">Aranha et al., 2020</xref>). The most recent version of NetMHCpan (NetMHCpan 4.1) is currently considered as the state-of-the-art in the MHC Class I-peptide binding prediction problem (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>).</p>
<p>NetMHCpan is a pan-specific model which predicts binding of peptides to any MHC molecule of known sequence using artificial neural networks. Since 2003, this model has gradually been improved and its last version for MHC Class I (NetMHCpan 4.1) has been introduced in 2020. This model is trained on a combination of the BA and EL peptide datasets where the inputs are sequences associated with MHC-peptide complexes (<xref ref-type="bibr" rid="B45">Tong, 2013</xref>). There are some specific features associated with this method that helps it to outperform other approaches: (i) instead of using the complete sequence of MHC molecules as input, NetMHCpan uses pseudo-sequences of MHC molecules with a fixed length (34 amino acids); these pseudo-sequences include those amino acids associated with the binding sites of the MHC molecule inferred from <italic>a priori</italic> knowledge; (ii) to accommodate peptides of different lengths (8&#x2013;15 in MHC Class I), the length is fixed to a uniform length of 9 <italic>via</italic> insertion and deletion of amino acids; (iii) additional features with specificity information of the peptides are used during the insertion and deletion steps; for example, the original length of the peptide is encoded as a categorical variable and the length of the sequence that was inserted/deleted is added as a different feature; (iv) NetMHCpan consists of several shallow neural networks and it implements the ensemble technique: using cross-validation, the training dataset is split into 5 parts and the model is trained five times, one for each split. Also, NetMHCpan uses shallow neural networks with one hidden layer containing 56 or 66 neurons that are trained using 10 different random initial weight configurations; thus, the ensemble NetMHCpan contains 100 different models.</p>
<p>As indicated above, the most recent NetMHCpan approach [version 4.1 (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>)] is based on shallow neural networks. In recent years, a number of more complex yet efficient methods such as deep neural networks have shown promising results in a number of fields (<xref ref-type="bibr" rid="B12">Deng et al., 2013</xref>; <xref ref-type="bibr" rid="B28">LeCun et al., 2015</xref>; <xref ref-type="bibr" rid="B27">Khan and Yairi, 2018</xref>; <xref ref-type="bibr" rid="B50">Voulodimos et al., 2018</xref>; <xref ref-type="bibr" rid="B23">Iuchi et al., 2021</xref>; <xref ref-type="bibr" rid="B33">Mohammadzadeh and Lejeune, 2021</xref>). For example, transformer models (<xref ref-type="bibr" rid="B47">Vaswani et al., 2017</xref>), a recent breakthrough in natural language processing, have shown that large models trained on unlabeled data are able to learn powerful representations of natural languages and can lead to significant improvements in many language modeling tasks (<xref ref-type="bibr" rid="B13">Devlin et al., 2018</xref>; <xref ref-type="bibr" rid="B21">Hu et al., 2022</xref>). Furthermore, it has been shown that collections of protein sequences can be treated as sentences so that similar techniques can be used to extract useful biological information from protein sequence databases (<xref ref-type="bibr" rid="B38">Rao et al., 2019</xref>; <xref ref-type="bibr" rid="B42">Rives et al., 2019</xref>). A highly successful example of this approach has been DeepMind&#x2019;s recent protein-folding method, using attention-based models (<xref ref-type="bibr" rid="B26">Jumper et al., 2020</xref>; <xref ref-type="bibr" rid="B29">Lensink et al., 2021</xref>; <xref ref-type="bibr" rid="B15">Egbert et al., 2021</xref>; <xref ref-type="bibr" rid="B19">Ghani et al., 2021</xref>). Currently, there are a number of publicly available pre-trained models which have been shown to be helpful in a variety of downstream protein related tasks (<xref ref-type="bibr" rid="B38">Rao et al., 2019</xref>; <xref ref-type="bibr" rid="B42">Rives et al., 2019</xref>; <xref ref-type="bibr" rid="B16">Elnaggar et al., 2020</xref>; <xref ref-type="bibr" rid="B40">Rao et al., 2020</xref>; <xref ref-type="bibr" rid="B39">Rao et al., 2021</xref>).</p>
<p>In the work reported in this paper, we consider using a number of such pre-trained models and Deep Learning (DL) methods to address the MHC Class I peptide binding prediction problem. One component of the approach in this work is based on transfer learning. In Deep Learning (DL), transfer learning is a method in which a DL model is first trained on a problem similar to the problem of interest; then, a portion or the whole of this pre-trained model is used for training the model of the desired problem. This approach is particularly advantageous when the amount of data for the problem of interest is limited, however, large databases associated with other problems with some similarity to the problem of interest exist (<xref ref-type="bibr" rid="B17">Fu and Bates, 2022</xref>). Fine-tuning a pre-trained model using the dataset associated with the problem of interest is one of the approaches in transfer learning and one that is used in this work. In this case, a portion, or all of the weights associated with the pre-trained model are used as the initial weights of a new DL model for the desired task. For example, in NLP, BERT (Bidirectional Encoder Representations from Transformers) is a pre-trained transformer model which is trained on a large corpus of unlabelled text including the entire Wikipedia (about 2,500 million words) and the Book Corpus (800 million words) (<xref ref-type="bibr" rid="B13">Devlin et al., 2018</xref>). Thereafter, the pre-trained model has been used for a number of NLP tasks such as text classification, text annotation, question answering, and language inference, to name a few.</p>
<p>Recently, following the successful results of pre-trained transformer models such as BERT and their transfer learning derivatives in NLP applications, similar approaches have been attempted in the protein field thanks to the substantial growth in the number of protein sequences. As a result, there are a number of pre-trained self-supervised BERT-like models applied to protein data in the form of unlabeled amino acid sequences which can be very useful for many protein task-specific problems using transfer learning (<xref ref-type="bibr" rid="B16">Elnaggar et al., 2020</xref>; <xref ref-type="bibr" rid="B40">Rao et al., 2020</xref>). Two recent works have considered using protein language models in the MHC-peptide binding problem. BERTMHC (<xref ref-type="bibr" rid="B11">Cheng et al., 2020</xref>) explores whether pre-trained protein sequence models can be helpful for MHC Class II-peptide binding prediction by focusing on algorithms that predict the likelihood of presentation of a peptide given a set of MHC Class II molecules. They show that models generated from transfer learning can achieve better performance on both binding and presentation prediction tasks compared to NetMHCIIpan4.0 (last version of NetMHCpan in MHC Class II (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>)). Another BERT-based model known as ImmunoBERT (<xref ref-type="bibr" rid="B18">Gasser et al., 2021</xref>) applies pre-trained transformer models in the MHC Class I-peptide binding problem. As reported, in this work they were not able to compare their model fairly with NetMHCPan (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>) and MHCflurry (<xref ref-type="bibr" rid="B35">O&#x2019;Donnell et al., 2020</xref>) performance due to a lack of access to the same training set. BERTMHC and ImmunoBERT both use the TAPE pre-trained models (<xref ref-type="bibr" rid="B38">Rao et al., 2019</xref>) which were trained on 31 million protein sequences, whereas now there are larger and more informative pre-trained models available such as ESM (<xref ref-type="bibr" rid="B40">Rao et al., 2020</xref>; <xref ref-type="bibr" rid="B30">Lin et al., 2022</xref>) and ProtTrans (<xref ref-type="bibr" rid="B16">Elnaggar et al., 2020</xref>) which are trained on more than 250 million protein sequences.</p>
<p>In the work reported in this paper, we focus on MHC Class I-peptide binding prediction and develop different approaches using the larger pre-trained protein language models. Two of these approaches are based on fine-tuning using a soft-max layer in one and a Graph Attention Network (GAT) in the other. Our third approach is based on a domain adaptation method to further pre-train the protein language models and enhance the fine-tuning performance. We evaluate the performance of our models using the standard metrics of the field and the same training and test sets as those of NetMHCpan 4.1. We show that our methods outperform NetMHCpan 4.1 over these test sets.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Methods</title>
<p>In this work, we considered two large protein language pre-trained models, ESM1b (<xref ref-type="bibr" rid="B40">Rao et al., 2020</xref>) and ESM2 (<xref ref-type="bibr" rid="B30">Lin et al., 2022</xref>), two BERT-based models which are trained on hundreds of millions of protein sequences. ESM1b is a pre-trained Transformer protein language model from Facebook AI Research which has been shown to outperform all tested single-sequence protein language models across a range of protein structure prediction tasks (<xref ref-type="bibr" rid="B40">Rao et al., 2020</xref>); its successor, ESM2, has achieved even better performance on protein folding related tasks. ESM1b and ESM2-650M have 33 layers with 650 million parameters and an embedding dimension of 1280, and the largest model we used, ESM2-3B, has 36 layers, embedding dimension of 2560 and 3 billion parameters. In our fine-tuning approaches, after including an additional layer at the end of the ESM models, we re-trained the entire set of parameters of ESM1b and ESM2 and trained the parameters of the added layer using the MHC-peptide dataset. Thus, the entire parameters, including the pre-trained weights of the model, were updated based on our dataset (<xref ref-type="fig" rid="F1">Figure 1</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Our fine-tuning architecture using NLP-based pre-trained models.</p>
</caption>
<graphic xlink:href="fbinf-03-1207380-g001.tif"/>
</fig>
<sec id="s2-1-1">
<title>2.1.1 ESM fine-tuning</title>
<p>Since ESM models can be regarded as transformer-based bidirectional language models (bi-LM), we borrowed an idea from a basic NLP task called Natural Language Inference (NLI) (<xref ref-type="bibr" rid="B8">Bowman et al., 2015</xref>) to perform MHC-peptide binding prediction. One of the NLI tasks is the sequence-pair classification problem, namely, predicting whether a text A (e.g., &#x201c;rabbits are herbivorous&#x201d;) can imply the semantics in a text B (e.g., &#x201c;rabbits do not eat rats&#x201d;). Similarly, in the MHC-peptide case, we would like to know whether a given peptide sequence (same as text A) binds to a given MHC sequence (same as text B), suggesting that applying an NLI-based model could be effective in MHC-peptide binding prediction. A common transformer-based NLI model combines text A and B into one sequence &#x201c;[BOS] seq-A [SEP] seq-B [EOS]&#x201d; as input, where [BOS], [SEP] and [EOS] are special tokens<xref ref-type="fn" rid="fn2">
<sup>1</sup>
</xref> in bi-LM vocabulary.</p>
<p>Suppose the amino acids in the MHC and the peptide sequences are <italic>M</italic>
<sub>1</sub>, &#x2026;, <italic>M</italic>
<sub>
<italic>p</italic>
</sub> and <italic>P</italic>
<sub>1</sub>, &#x2026;, <italic>P</italic>
<sub>
<italic>q</italic>
</sub>, respectively. We generate the sequence &#x201c;[BOS], <italic>M</italic>
<sub>1</sub>, &#x2026;, <italic>M</italic>
<sub>
<italic>p</italic>
</sub>, [SEP], <italic>P</italic>
<sub>1</sub>, &#x2026;, <italic>P</italic>
<sub>
<italic>q</italic>
</sub>, [EOS]&#x201d; with length <italic>p</italic> &#x2b; <italic>q</italic> &#x2b; 3 as the ESM model input, and obtain the same size embedding vectors <inline-formula id="inf1">
<mml:math id="m1">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">BOS</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">SEP</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">EOS</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> from the last layer of ESM models, corresponding to the special tokens and the amino acids in the MHC and the peptide. As a common strategy in NLP sequence classification tasks, we use the embedding of [BOS] to be the MHC-peptide sequence-pair embedding vector <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B22">Ibtehaz and Kihara, 2023</xref>). Finally, passing <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> through a softmax classifier layer, we output the probability of binding and use it to compute the loss and apply back-propagation. Compared to embedding the MHC and the peptide separately, this compound input allows the transformer to use the attention mechanism to further extract the interactive information between the amino acids in the MHC and the peptide, thus, helping the binding prediction.</p>
<p>Although ESM models are well pre-trained in an unsupervised manner, using a large number of universal sequences, we know that MHCs are highly specific types of protein sequences, so the embedding from the pre-trained ESM models may not be optimal for the specific MHC task and input format. Therefore, we not only need to train the final softmax classifier but need to train the ESM model parameters as well to improve the sequence-pair embedding. We applied a fine-tuning which is commonly used in NLP. Initialized from the pre-trained ESM model parameters, we updated the parameters in the whole network using a small learning rate during the back-propagation, so that valuable information in the pre-trained ESM models is maintained while the fine-tuned ESM models provided a more informative embedding specific to the MHC tasks.</p>
</sec>
<sec id="s2-1-2">
<title>2.1.2 ESM domain adaptation</title>
<p>In NLP, domain adaptation pre-training is an important tool to introduce domain-specific information into a bi-LM. A BERT model pre-trained on general corpora (<xref ref-type="bibr" rid="B13">Devlin et al., 2018</xref>) (e.g., Wikipedia) can be further pre-trained by the same masked language modeling (MLM) methods but using corpora from specific domains such as clinical medicine (<xref ref-type="bibr" rid="B2">Alsentzer et al., 2019</xref>) in order to gain better down-stream task performance in different knowledge domains. This idea fits our protein language models as well because ESM models were pre-trained on general full protein sequences whereas our MHC-peptide binding prediction focuses on MHC pseudo-sequences and short peptides, which were not available in the ESM pre-training data. Therefore, we applied domain adaptation to the ESM models in order to offer the ESM models more knowledge about the MHC pseudo-sequences and the peptides.</p>
<p>We still use the NetMHCpan V4.1 training set as our domain adaptation pre-training set. For an MHC-peptide pair &#x201c;[BOS], <italic>M</italic>
<sub>1</sub>, &#x2026;, <italic>M</italic>
<sub>
<italic>p</italic>
</sub>, [SEP], <italic>P</italic>
<sub>1</sub>, &#x2026;, <italic>P</italic>
<sub>
<italic>q</italic>
</sub>, [EOS]&#x201d;, we first randomly mask 7 amino acids (around 15%), and then feed this masked sequence pair to the pre-trained ESM models. Note that with a probability of 0.8, an amino acid to be masked will be masked by a special token [MASK], otherwise it will be &#x201c;masked&#x201d; by the original amino acid, which resembles the MLM setting in BERT. The ESM models will then exploit the information from the visible context of amino acids, and finally use a classification head to predict the masked amino acids. As a result, the special structural characteristics of the MHC pseudo-sequences and the peptides will be further learned, and our domain-adapted ESM models can better fit the MHC-related tasks, compared with the vanilla ESM models. For one MHC-peptide pair, the loss to be minimized is the mean cross-entropy loss between the predicted and the ground truth masked amino acids. During the ESM domain adaptation pre-training, we still update all parameters of the ESM models, and our domain-adapted ESM models will be used as the initialization of the MHC-peptide binding prediction fine-tuning task described in the previous section.</p>
</sec>
<sec id="s2-1-3">
<title>2.1.3 ESM-GAT fine-tuning</title>
<p>Here, we consider our second approach to fine-tuning. Molecular structure-based biological data such as proteins, can be modeled with graph structures in which amino-acids or atoms are considered as nodes, and contacts or bonds are considered as edges. It has been shown that Graph Neural Networks (GNNs), as a branch of deep learning in non-Euclidean spaces, perform well in various applications in bioinformatics (<xref ref-type="bibr" rid="B53">Zhang et al., 2021</xref>). In our context, the interaction between an MHC and a peptide can be described by a graph in which the amino-acids are considered as the nodes and the interaction between them as edges. To model such a graph information, we added a variant model of GNN known as Graph Attention Network (GAT) as the last layer of the ESM network. GAT is a novel neural network architecture that operates on graph-structured data by leveraging attention layers to address the shortcomings of prior methods based on graph convolutions or their approximations (<xref ref-type="bibr" rid="B48">Veli&#x10d;kovi&#x107; et al., 2017</xref>). For each MHC-peptide pair, we used a directed graph <inline-formula id="inf4">
<mml:math id="m4">
<mml:mi mathvariant="script">G</mml:mi>
</mml:math>
</inline-formula>, where the nodes <italic>N</italic>
<sub>1</sub>, &#x2026;, <italic>N</italic>
<sub>
<italic>p</italic>&#x2b;<italic>q</italic>&#x2b;3</sub> represent the <italic>p</italic> &#x2b; <italic>q</italic> &#x2b; 3 amino acids and the special tokens as described above, and an edge (<italic>N</italic>
<sub>
<italic>i</italic>
</sub>, <italic>N</italic>
<sub>
<italic>j</italic>
</sub>) indicates that amino acids <italic>i</italic> and <italic>j</italic> are in contact with each other. Denote the neighbor set of amino acid <italic>i</italic> as <inline-formula id="inf5">
<mml:math id="m5">
<mml:mi mathvariant="script">A</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>:</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">G</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>; then, each embedding vector <bold>v</bold>
<sub>
<italic>i</italic>
</sub> is updated as a weighted average of its transformed neighbor embedding vectors:<disp-formula id="equ1">
<mml:math id="m6">
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="bold">W</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:math>
</disp-formula>where <bold>W</bold> is a weight matrix for vector transformation, and the weight <italic>&#x3b1;</italic>
<sub>
<italic>ij</italic>
</sub> is computed using an attention mechanism. Suppose <bold>z</bold>
<sub>
<italic>ij</italic>
</sub> is the concatenation of vectors <bold>Wv</bold>
<sub>
<italic>i</italic>
</sub> and <bold>Wv</bold>
<sub>
<italic>j</italic>
</sub> and <bold>c</bold> is a parameter vector, then the weight <italic>&#x3b1;</italic>
<sub>
<italic>ij</italic>
</sub> is given by:<disp-formula id="equ2">
<mml:math id="m7">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo stretchy="false">&#x2329;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x232a;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo stretchy="false">&#x2329;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x232a;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:math>
</disp-formula>where <italic>&#x3c3;</italic> is an activation function. Note that the attention mechanism here is known as <italic>additive</italic> attention, which is different from the dot-product attention mechanism used in ESM and other transformer-based models.</p>
<p>After each GAT layer, we update the embedding vector for the amino acids and the special tokens as <inline-formula id="inf6">
<mml:math id="m8">
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">BOS</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">SEP</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">EOS</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>, and more GAT layers follow. Here, in our implementation, we use two fully connected GAT layers. Same as vanilla transformer model (<xref ref-type="bibr" rid="B47">Vaswani et al., 2017</xref>), we apply multi-head attention mechanism in which for each GAT layer, we split the parameters and pass each split independently through a separate head. Particularly, in the first GAT layer we use 8 attention heads which are then concatenated together and passed to the next layer while in the final GAT layer we average the heads of a certain token. We finally use the embedding vector of [BOS] in the final GAT layer as the MHC-peptide sequence pair embedding vector to determine the binding prediction. The final GAT layer was meant to use the attention mechanism to aggregate all the node information into [BOS] position by letting [BOS] token contact with all the amino acids in the graphs which makes the [BOS] embedding potentially a more powerful sequence embedding than simply using the average of the embedding vectors output by the first GAT layer. Compared to using only ESM dot-product attention layers and a linear classification head, now we are adding more GAT additive attention layers to dynamically refine the ESM embedding and enhance the final binding classification.</p>
<p>Note that the contact information can be defined differently through graphs. If in the absence of specific information about the contacts, fully-connected graphs are used as we did, the dependency among any amino acids can be further exploited by those additive attention layers, similar to the ESM layers. However, if prior information on contacts is available and is represented in the graphs, such information can also be introduced to the GAT layers by allowing the additive attention mechanism to happen only between the desired amino acids.</p>
</sec>
</sec>
<sec id="s2-2">
<title>2.2 Dataset</title>
<sec id="s2-2-1">
<title>2.2.1 Training set</title>
<p>We used the training set used by the last version of NetMHCpan (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>), including 13 millions binary labeled MHC-peptide binding samples, generated from two main data sources: (i) the BA peptides derived from in-vitro Peptide-MHC binding assays, and (ii) the EL peptides derived from mass spectrometry experiments. However, it has been shown that the results from the mass spectrometry EL experiments are mostly poly-specific, i.e., they contain ligands matching multiple binding motifs (<xref ref-type="bibr" rid="B3">Alvarez et al., 2019</xref>). That being said, for most of the samples in the EL dataset, each peptide is associated with multiple alleles (from 2 to 6 alleles for each peptide). Thus, in this training set, the EL dataset is composed of two subsets: (i): Single-Allele (SA) peptides assigned to single MHCs and (ii) Multi-Allele (MA) peptides with multiple MHC options to be assigned. <xref ref-type="table" rid="T1">Table 1</xref> shows the distribution of the aforementioned dataset which indicates that more than 67% of the dataset is associated with EL-MA. According to (<xref ref-type="bibr" rid="B3">Alvarez et al., 2019</xref>), the existence of the MA dataset introduces some challenges in terms of data analysis and interpretation; therefore, to train a binary MHC-peptide predictor, a process, known as deconvoluting the MA binding motifs, is needed to convert these EL-MA data to a single peptide-MHC pair (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Distribution of training set used in NetMHCpan 4.1 (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>); Columns correspond to each type of training data, for which the number of positive and negative samples, and the total amount of unique MHCs are shown. A threshold of 500&#xa0;nM is used to define positive BA data points.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="3" align="left">Binding affinity</th>
<th colspan="3" align="left">EL (single allele)</th>
<th colspan="3" align="left">EL (multi allele)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Positives</td>
<td align="left">Negatives</td>
<td align="left">MHCs</td>
<td align="left">Positives</td>
<td align="left">Negatives</td>
<td align="left">MHCs</td>
<td align="left">Positives</td>
<td align="left">Negatives</td>
<td align="left">MHCs</td>
</tr>
<tr>
<td align="left">52,402</td>
<td align="left">155,691</td>
<td align="left">170</td>
<td align="left">218,962</td>
<td align="left">3,813,877</td>
<td align="left">142</td>
<td align="left">446,530</td>
<td align="left">8,395,021</td>
<td align="left">112</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Deconvolution of multi allelic (MA) data</title>
<p>To deconvolute the EL-MA dataset, several computational approaches have been used based on unsupervised sequence clustering (<xref ref-type="bibr" rid="B6">Bassani-Sternberg and Gfeller, 2016</xref>; <xref ref-type="bibr" rid="B5">Bassani-Sternberg et al., 2017</xref>). Although these methods show some progress in dealing with the MA dataset, they have some shortcomings; for example, they do not work in cell lines including MHC alleles with similar binding motifs. Therefore, in the new version of NetMHCPan (Version 4.1), they present a new framework, NNAlign-MA (<xref ref-type="bibr" rid="B3">Alvarez et al., 2019</xref>), which works better than the previous approaches. NNAlign-MA is a neural network framework, which is able to deconvolute the MA dataset during the training of the MHC-peptide binding predictor. Recently (<xref ref-type="bibr" rid="B11">Cheng et al., 2020</xref>), attempted to solve this problem in MHC Class II by using a multiple instance learning (MIL) framework. MIL is a supervised machine learning approach, where the task is to learn from data including positive and negative bags of instances. Each bag may contain many instances and a bag is labeled positive if at least one instance in it is positive (<xref ref-type="bibr" rid="B32">Maron and Lozano-P&#xe9;rez, 1998</xref>). Assume the <italic>i</italic>-th bag includes m alleles as <italic>A</italic>
<sub>
<italic>i</italic>
</sub> &#x3d; {<italic>a</italic>
<sub>
<italic>i</italic>1</sub>, <italic>a</italic>
<sub>
<italic>i</italic>2</sub>, &#x2026;, <italic>a</italic>
<sub>
<italic>im</italic>
</sub>} which is associated with peptide sequence <italic>s</italic>
<sub>
<italic>i</italic>
</sub>. At each training epoch, for each instance in the <italic>i</italic>-th bag, <italic>x</italic>
<sub>
<italic>ij</italic>
</sub> &#x3d; (<italic>a</italic>
<sub>
<italic>ij</italic>
</sub>, <italic>s</italic>
<sub>
<italic>i</italic>
</sub>), the probability of whether that instance is positive, <italic>p</italic>(<italic>y</italic>
<sub>
<italic>ij</italic>
</sub> &#x3d; 1&#x7c;<italic>x</italic>
<sub>
<italic>ij</italic>
</sub>) is defined as <inline-formula id="inf7">
<mml:math id="m9">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> where <italic>f</italic>
<sub>
<italic>&#x3b8;</italic>
</sub> is the neural network model; in (<xref ref-type="bibr" rid="B11">Cheng et al., 2020</xref>) max pooling is used as a symmetric pooling operator to calculate the prediction of the bag from the predictions of instances within it. Here, in our work, we follow this MIL idea to deal with the EL-MA dataset.</p>
</sec>
<sec id="s2-2-3">
<title>2.2.3 Test set</title>
<p>In order to have a fair comparison of our model and NetMHCPan 4.1, we used the same test set as provided in their work (available in the <xref ref-type="sec" rid="s10">Supplementary Section</xref>). This dataset is associated with a collection of 36&#xa0;EL-SA datasets, downloaded from (<xref ref-type="bibr" rid="B1">Abelin et al., 2017</xref>). Each dataset is well enriched, length-wise, with a number of negative decoy peptides equal to 5 times the number of ligands of the most abundant peptide length.</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Metric</title>
<p>Predicting the binding affinity of MHC with a peptide is a binary classification problem. Typical metrics for assessing the quality of binary classification models for a given task include precision, accuracy, recall, receiver operating characteristic curve (ROC) and the corresponding Area Under the Curve (AUC). In this work, we use AUC-ROC and a specific precision metric known as positive predictive value (PPV); AUC and PPV have been used as the main metrics in previous works in MHC-peptide binding prediction (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>; <xref ref-type="bibr" rid="B35">O&#x2019;Donnell et al., 2020</xref>). AUC is an evaluation metric for binary classification problems which measures the area under the ROC curve. AUC ranges in value from 0 to 1 and models with higher AUC perform better at distinguishing between the positive and negative classes. PPV is another metric which specifically is defined in this area and is interpretable as a model&#x2019;s ability to rank positive samples far above the negative samples. PPV is defined as fraction of true positive samples (hits) among the top-scoring <inline-formula id="inf8">
<mml:math id="m10">
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:math>
</inline-formula> fraction of samples, assuming that the ratio of the number of positive samples to negatives (decoys) is 1: N (known as hit-decoy ratio). Since NetMHCpan (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>) uses hit-ratio 1:19 and MHCflurry (<xref ref-type="bibr" rid="B35">O&#x2019;Donnell et al., 2020</xref>) uses hit-ratio 1:99, here in this work, we use both.</p>
<p>Beyond AUC-ROC and PPV, we also consider three more metrics: F1 score, Precision-Recall Area Under Curve (AUC-PR), and Matthews Correlation Coefficient (MMC). These metrics provide a comprehensive evaluation of the model&#x2019;s performance by measuring the balance between precision and recall, and summarizing performance on imbalanced datasets.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<p>In order to evaluate and compare the performance of our approaches with the state-of-the-art method, we used the latest version of NetMHCpan server (Version 4.1); as mentioned above, the same training and test sets from (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>) were used in this study. The list of independent EL SA test set including the MHC molecules, the number of peptides and the distribution of positives and negatives for each case is provided in the <xref ref-type="sec" rid="s10">Supplementary Material</xref>.</p>
<p>To arrive at the hit-decoy ratios of 1:19 and 1:99 for each case, we followed a random sampling approach that was repeated 1000 times. As a result, for each MHC molecule, the PPV values are sample averages of 1000 values. Additionally, in <xref ref-type="fig" rid="F2">Figure 2</xref> we provide a comparison over a range of hit-decoy ratios.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>PPV Comparison of ESM2 model vs NetMHCpan 4.1 over different hit-decoy ratio.</p>
</caption>
<graphic xlink:href="fbinf-03-1207380-g002.tif"/>
</fig>
<p>To present the results of the comparison of our fine-tuning as well as our domain adaptation approaches with NetMHCpan, we provide two figures for each hit-decoy ratio in what follows: (a) a bar plot that provides a comparison of PPVs of our approach and NetMHCpan for each MHC molecule in the test set, and (b) a scatter plot of the same PPV values that provides a better visual summary of performance comparison.</p>
<p>Since there was not a significant difference in performance when using the ESM1b, ESM2-650M, or ESM2-3B, we report the ESM2-3B values in this section which were slightly better in mean performance than others. <xref ref-type="table" rid="T2">Tables 2</xref>, <xref ref-type="table" rid="T3">3</xref> below show the summary of the results for fine-tuning and domain adaptation which provides the mean of using PPV, AUC-ROC, F1, AUC-PR, and MMC averages over all MHC molecules in the test set.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Summary table of comparison of the mean of our models and NetMHCpan (V4.1) AUC-ROC and PPV over all test sets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Models</th>
<th align="center">PPV (hit-decoy ratio: 1:19)</th>
<th align="center">PPV (hit-decoy ratio: 1:99)</th>
<th align="center">AUC-ROC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">NetMHCpan4_ 1</td>
<td align="right">0.791</td>
<td align="right">0.671</td>
<td align="right">0.950</td>
</tr>
<tr>
<td align="left">ESM1b</td>
<td align="right">0.834</td>
<td align="right">0.737</td>
<td align="right">0.977</td>
</tr>
<tr>
<td align="left">ESM2_650M</td>
<td align="right">0.837</td>
<td align="right">0.742</td>
<td align="right">0.976</td>
</tr>
<tr>
<td align="left">ESM2_3B</td>
<td align="right">0.844</td>
<td align="right">0.753</td>
<td align="right">0.976</td>
</tr>
<tr>
<td align="left">ESM1b_domain</td>
<td align="right">0.851</td>
<td align="right">0.756</td>
<td align="right">0.979</td>
</tr>
<tr>
<td align="left">ESM2_650M_domain</td>
<td align="right">0.850</td>
<td align="right">0.756</td>
<td align="right">0.980</td>
</tr>
<tr>
<td align="left">ESM2_3B_domain</td>
<td align="right">0.857</td>
<td align="right">0.769</td>
<td align="right">0.981</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Summary table of comparison of the mean of our models and NetMHCpan (V4.1) F1, AUC-PR, and MMC over all test sets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Models</th>
<th align="right">Fl</th>
<th align="right">AUC-PR</th>
<th align="right">MMC</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">NetMHCpan4_l</td>
<td align="right">0.711</td>
<td align="right">0.833</td>
<td align="right">0.726</td>
</tr>
<tr>
<td align="left">ESM1b</td>
<td align="right">0.771</td>
<td align="right">0.885</td>
<td align="right">0.779</td>
</tr>
<tr>
<td align="left">ESM2 650M</td>
<td align="right">0.785</td>
<td align="right">0.888</td>
<td align="right">0.791</td>
</tr>
<tr>
<td align="left">ESM2 3B</td>
<td align="right">0.788</td>
<td align="right">0.893</td>
<td align="right">0.794</td>
</tr>
<tr>
<td align="left">ESM1b_domain</td>
<td align="right">0.794</td>
<td align="right">0.902</td>
<td align="right">0.799</td>
</tr>
<tr>
<td align="left">ESM2_650M_domain</td>
<td align="right">0.796</td>
<td align="right">0.900</td>
<td align="right">0.801</td>
</tr>
<tr>
<td align="left">ESM2 3B domain</td>
<td align="right">0.801</td>
<td align="right">0.908</td>
<td align="right">0.806</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3-1">
<title>3.1 ESM fine-tuning</title>
<p>As seen in <xref ref-type="fig" rid="F3">Figure 3</xref> our fine-tuning method outperforms NetMHCpan over all hit-decoy ratios in the 35 different test sets; only for HL-B18:01, at ratio 1:19, NetMHCpan performs slightly better. Also, as seen in <xref ref-type="fig" rid="F4">Figure 4</xref>, at ratio 1:99 the model outperforms NetMHCpan for all 36 test set including the HL-B18:01.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>PPV Comparison (hit-decoy ratio: 1:19) of our fine-tuning method with the latest NetMHCpan server (Version 4.1) over the same training and test sets (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>). <bold>(A)</bold> Bar plots associated with each test set. <bold>(B)</bold> Scatter plot: each point is the PPV of each group of test set.</p>
</caption>
<graphic xlink:href="fbinf-03-1207380-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>PPV Comparison (hit-decoy ratio: 1:99) of our fine-tuning method with the latest NetMHCpan server (Version 4.1) over the same training and test sets (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>). <bold>(A)</bold> Bar plots associated with each test set. <bold>(B)</bold> Scatter plot: each point is the PPV of each group of test set.</p>
</caption>
<graphic xlink:href="fbinf-03-1207380-g004.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 ESM domain adaptation</title>
<p>
<xref ref-type="fig" rid="F5">Figures 5</xref>, <xref ref-type="fig" rid="F6">6</xref> show that our domain adaptation model outperforms NetMHCpan over all hit-decoy ratios in the 35 different test sets; only for HL-B18:01, at ratio 1:19, NetMHCpan slightly performs better. In addition, the performance of the domain adaptation approach is slightly better than the fine-tuning approach.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>PPV Comparison (hit-decoy ratio: 1:19) of our domain-adaptation method with the latest NetMHCpan server (Version 4.1) over the same training and test sets (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>). <bold>(A)</bold> Bar plots associated with each test set. <bold>(B)</bold> Scatter plot: each point is the PPV of each group of test set.</p>
</caption>
<graphic xlink:href="fbinf-03-1207380-g005.tif"/>
</fig>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>PPV Comparison (hit-decoy ratio: 1:99) of our domain-adaptation method with the latest NetMHCpan server (Version 4.1) over the same training and test sets (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>). <bold>(A)</bold> Bar plots associated with each test set. <bold>(B)</bold> Scatter plot: each point is the PPV of each group of test set.</p>
</caption>
<graphic xlink:href="fbinf-03-1207380-g006.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 ESM-GAT fine-tuning</title>
<p>Given the superior performance of ESM fine-tuning in comparison with NetMHCpan, to assess the performance of ESM-GAT fine-tuning, we compared its performance with that of ESM fine-tuning. In this case, a hit-decoy ratios of 1:19 was considered. We found that in the case where we subdivided the training and test sets between peptides of length 8 and 9 on the one hand and peptides of size 10&#x2013;15 on the other, ESM-GAT fine-tuning outperformed ESM fine-tuning. Specifically, we used subsets of the training set that included samples associated with peptides of length 8 and 9 and compared both methods over two test sets. As can be seen in <xref ref-type="fig" rid="F7">Figure 7</xref>, ESM-GAT outperformed ESM fine-tuning when the test set with peptide length 10&#x2013;15 was considered (red dots), while the results were almost the same when using the test set with peptides of length 8 and 9 (blue dots). Bar plots associated with these figures are available in the <xref ref-type="sec" rid="s10">Supplementary Section</xref>. This observation suggests that GAT has the potential to improve the ability of the model to predict binding of peptides with lengths different from those considered in the training set. The testing of this conjecture is a subject of future research.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>ESM-GAT fine-tuning outperforms the ESM fine-tuning method when the test set with peptide length 10&#x2013;15 is considered (red points) while the results are almost the same when using the test set with peptides of length 8 and 9 (blue points).</p>
</caption>
<graphic xlink:href="fbinf-03-1207380-g007.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>Predicting peptides that bind to the major histocompatibility complex (MHC) Class I is an important problem in studying the immune system response and a plethora of approaches have been developed to tackle this problem. NetMHCpan 4.1 is developed based on training shallow neural networks (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>) and is currently considered the state-of-the-art for MHC Class I-peptide binding prediction. A number of recent works have focused on using protein language models in MHC-peptide binding problems. Protein language models developed based on deep learning approaches, such as attention-based transformer models, have shown significant progress towards solving a number of challenging problems in biology, most importantly, the protein structure prediction problem (<xref ref-type="bibr" rid="B25">Jumper et al., 2021</xref>). BERTMHC (<xref ref-type="bibr" rid="B11">Cheng et al., 2020</xref>) and ImmunoBERT (<xref ref-type="bibr" rid="B18">Gasser et al., 2021</xref>) for the first time applied the pre-trained protein language models in MHC-peptide binding problems. Both methods used a relatively small pre-trained model ((<xref ref-type="bibr" rid="B38">Rao et al., 2019</xref>) was trained with 31 million protein sequences); currently, there are substantially larger and more informative models such as ESM1b (<xref ref-type="bibr" rid="B40">Rao et al., 2020</xref>) and ProtTrans (<xref ref-type="bibr" rid="B16">Elnaggar et al., 2020</xref>) which are trained on more than 250 million protein sequences. In the work reported in this paper we focus on MHC Class I peptide binding prediction by developing approaches based on large pre-trained protein language models, ESM1b (<xref ref-type="bibr" rid="B40">Rao et al., 2020</xref>) and ESM2 (<xref ref-type="bibr" rid="B30">Lin et al., 2022</xref>). We follow two fine-tuning approaches using a soft-max layer and Graph Attention Network (GAT) as well as implement a domain adaptation pre-training for ESM models. In order to have a fair comparison, we train our model using the same training set used by NetMHCpan 4.1 (<xref ref-type="bibr" rid="B41">Reynisson et al., 2020</xref>) and evaluate our model using the same test set. We show, using the standard performance metrics in this area, that our model outperforms NetMHCpan. As reported in the paper, adding Graph Attention Network (GAT) to the ESM networks, improved the ability of the model to predict peptides with lengths different from those considered in the training set; this feature is expected to be beneficial for training models beyond MHC Type I.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://services.healthtech.dtu.dk/suppl/immunology/NAR_NetMHCpan_NetMHCIIpan/">https://services.healthtech.dtu.dk/suppl/immunology/NAR_NetMHCpan_NetMHCIIpan/</ext-link>.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>DK, NH, PV, and IP designed research; NH, BH, and MI performed research; NH, BH, IP, PV, SV, and DK analyzed data; PV, NH, and BH wrote the paper, NH, BH, IP, PV, SV, and DK reviewed and edited the paper. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This work was supported in part by the National Institutes of Health grants R01 GM135930, RM1135136, R35GM118078, and R01GM140098, by the Boston University Clinical and Translational Science Award (CTSA) under NIH/NCATS grant UL54 TR004130; by the National Science Foundation grants IIS-1914792, DMS-1664644, DMS-2054251, and CNS-1645681; and by the Office of Naval Research grant N00014-19-1-2571.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fbinf.2023.1207380/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fbinf.2023.1207380/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn2">
<label>1</label>
<p>A token is a string of contiguous characters between two spaces, or between a space and punctuation marks.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abelin</surname>
<given-names>J. G.</given-names>
</name>
<name>
<surname>Keskin</surname>
<given-names>D. B.</given-names>
</name>
<name>
<surname>Sarkizova</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hartigan</surname>
<given-names>C. R.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Sidney</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Mass spectrometry profiling of hla-associated peptidomes in mono-allelic cells enables more accurate epitope prediction</article-title>. <source>Immunity</source> <volume>46</volume>, <fpage>315</fpage>&#x2013;<lpage>326</lpage>. <pub-id pub-id-type="doi">10.1016/j.immuni.2017.02.007</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alsentzer</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Murphy</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Boag</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>W.-H.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Naumann</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Publicly available clinical bert embeddings</article-title>. <comment>arXiv preprint arXiv:1904.03323</comment>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alvarez</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Reynisson</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Barra</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Buus</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ternette</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Connelley</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Nnalign_ma; mhc peptidome deconvolution for accurate mhc binding motif characterization and improved t-cell epitope predictions</article-title>. <source>Mol. Cell. Proteomics</source> <volume>18</volume>, <fpage>2459</fpage>&#x2013;<lpage>2477</lpage>. <pub-id pub-id-type="doi">10.1074/mcp.tir119.001658</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aranha</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Jewel</surname>
<given-names>Y. S.</given-names>
</name>
<name>
<surname>Beckman</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Weiner</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Mitchell</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Parks</surname>
<given-names>J. M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Combining three-dimensional modeling with artificial intelligence to increase specificity and precision in peptide&#x2013;mhc binding predictions</article-title>. <source>J. Immunol.</source> <volume>205</volume>, <fpage>1962</fpage>&#x2013;<lpage>1977</lpage>. <pub-id pub-id-type="doi">10.4049/jimmunol.1900918</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bassani-Sternberg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chong</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Guillaume</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Solleder</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pak</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gannon</surname>
<given-names>P. O.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Deciphering hla-i motifs across hla peptidomes improves neo-antigen predictions and identifies allostery regulating hla specificity</article-title>. <source>PLoS Comput. Biol.</source> <volume>13</volume>, <fpage>e1005725</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005725</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bassani-Sternberg</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gfeller</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Unsupervised hla peptidome deconvolution improves ligand prediction accuracy and predicts cooperative effects in peptide&#x2013;hla interactions</article-title>. <source>J. Immunol.</source> <volume>197</volume>, <fpage>2492</fpage>&#x2013;<lpage>2499</lpage>. <pub-id pub-id-type="doi">10.4049/jimmunol.1600808</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boehm</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Bhinder</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Raja</surname>
<given-names>V. J.</given-names>
</name>
<name>
<surname>Dephoure</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Elemento</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Predicting peptide presentation by major histocompatibility complex class i: an improved machine learning approach to the immunopeptidome</article-title>. <source>BMC Bioinforma.</source> <volume>20</volume>, <fpage>7</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2561-z</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bowman</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Angeli</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Potts</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Manning</surname>
<given-names>C. D.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>A large annotated corpus for learning natural language inference</article-title>. <comment>arXiv preprint arXiv:1508.05326</comment>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bui</surname>
<given-names>H.-H.</given-names>
</name>
<name>
<surname>Sidney</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sathiamurthy</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sinichi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Purton</surname>
<given-names>K.-A.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>Automated generation and evaluation of specific mhc binding predictive tools: arb matrix applications</article-title>. <source>Immunogenetics</source> <volume>57</volume>, <fpage>304</fpage>&#x2013;<lpage>314</lpage>. <pub-id pub-id-type="doi">10.1007/s00251-005-0798-y</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Caron</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Kowalewski</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Koh</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Sturm</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Schuster</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Aebersold</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Analysis of major histocompatibility complex (mhc) immunopeptidomes using mass spectrometry</article-title>. <source>Mol. Cell. Proteomics</source> <volume>14</volume>, <fpage>3105</fpage>&#x2013;<lpage>3117</lpage>. <pub-id pub-id-type="doi">10.1074/mcp.o115.052431</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bendjama</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Rittner</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Malone</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Bertmhc: improves mhc-peptide class ii interaction prediction with transformer and multiple instance learning</article-title>. <source>bioRxiv</source>.</citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Deng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.-T.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Seide</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). &#x201c;<article-title>Recent advances in deep learning for speech research at microsoft</article-title>,&#x201d; in <source>2013 IEEE international conference on acoustics, speech and signal processing</source> (<publisher-name>IEEE</publisher-name>), <fpage>8604</fpage>&#x2013;<lpage>8608</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Devlin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.-W.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Toutanova</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Bert: pre-training of deep bidirectional transformers for language understanding</article-title>. <comment>arXiv preprint arXiv:1810.04805</comment>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Doytchinova</surname>
<given-names>I. A.</given-names>
</name>
<name>
<surname>Flower</surname>
<given-names>D. R.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Toward the quantitative prediction of t-cell epitopes: comfa and comsia studies of peptides with affinity for the class i mhc molecule hla-a&#x2a; 0201</article-title>. <source>J. Med. Chem.</source> <volume>44</volume>, <fpage>3572</fpage>&#x2013;<lpage>3581</lpage>. <pub-id pub-id-type="doi">10.1021/jm010021j</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Egbert</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ghani</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Ashizawa</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kotelnikov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Desta</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Assessing the binding properties of casp14 targets and models</article-title>. <source>Proteins Struct. Funct. Bioinforma.</source> <volume>89</volume>, <fpage>1922</fpage>&#x2013;<lpage>1939</lpage>. <pub-id pub-id-type="doi">10.1002/prot.26209</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elnaggar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Heinzinger</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dallago</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Rihawi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Prottrans: towards cracking the language of life&#x2019;s code through self-supervised deep learning and high performance computing</article-title>. <comment>arXiv preprint arXiv:2007.06225</comment>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Bates</surname>
<given-names>P. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Application of deep learning methods: from molecular modelling to patient classification</article-title>. <source>Exp. Cell. Res.</source> <volume>418</volume>, <fpage>113278</fpage>. <pub-id pub-id-type="doi">10.1016/j.yexcr.2022.113278</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gasser</surname>
<given-names>H.-C.</given-names>
</name>
<name>
<surname>Bedran</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Goodlett</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Alfaro</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rajan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Interpreting bert architecture predictions for peptide presentation by mhc class i proteins</article-title>. <comment>arXiv preprint arXiv:2111.07137</comment>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghani</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Desta</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Jindal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kotelnikov</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Improved docking of protein models by a combination of alphafold2 and cluspro</article-title>. <source>bioRxiv</source>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grebenkin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gaivoronsky</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Kazyonnov</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kulagin</surname>
<given-names>a.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Application of an ensemble of neural networks and methods of statistical mechanics to predict binding of a peptide to a major histocompatibility complex</article-title>. <source>Comput. Res. Model</source>.</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hosseini</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Parolin</surname>
<given-names>E. S.</given-names>
</name>
<name>
<surname>Osorio</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Brandt</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). &#x201c;<article-title>Conflibert: a pre-trained language model for political conflict and violence</article-title>,&#x201d; in <source>Proceedings of the 2022 conference of the north American chapter of the association for computational linguistics: human language technologies</source>, <fpage>5469</fpage>&#x2013;<lpage>5482</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ibtehaz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kihara</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Application of sequence embedding in protein sequence-based predictions</article-title>,&#x201d; in <source>Machine learning in bioinformatics of protein sequences: algorithms, databases and resources for modern protein bioinformatics</source> (<publisher-name>World Scientific</publisher-name>), <fpage>31</fpage>&#x2013;<lpage>55</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iuchi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Matsutani</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yamada</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Iwano</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Sumi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hosoda</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Representation learning applications in biological sequence analysis</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>19</volume>, <fpage>3198</fpage>&#x2013;<lpage>3208</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2021.05.039</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Janeway</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Travers</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Walport</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Capra</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2001</year>). <source>Immunobiology</source>. <publisher-name>Taylor &#x26; Francis Group UK: Garland Science</publisher-name>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jumper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Figurnov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ronneberger</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Highly accurate protein structure prediction with alphafold</article-title>. <source>Nature</source> <volume>596</volume>, <fpage>583</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jumper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Figurnov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tunyasuvunakool</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>High accuracy protein structure prediction using deep learning</article-title>. <source>Fourteenth Crit. Assess. Tech. Protein Struct. Predict.</source> <volume>22</volume>, <fpage>24</fpage>. <comment>Abstract Book</comment>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yairi</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A review on the application of deep learning in system health management</article-title>. <source>Mech. Syst. Signal Process.</source> <volume>107</volume>, <fpage>241</fpage>&#x2013;<lpage>265</lpage>. <pub-id pub-id-type="doi">10.1016/j.ymssp.2017.11.024</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>nature</source> <volume>521</volume>, <fpage>436</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lensink</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Brysbaert</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Mauri</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Nadzirin</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Velankar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chaleil</surname>
<given-names>R. A.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Prediction of protein assemblies, the next frontier: the casp14-capri experiment</article-title>. <source>Proteins Struct. Funct. Bioinforma.</source> <volume>89</volume>, <fpage>1800</fpage>&#x2013;<lpage>1823</lpage>. <pub-id pub-id-type="doi">10.1002/prot.26222</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Akin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Rao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hie</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Evolutionary-scale prediction of atomic level protein structure with a language model</article-title>. <source>bioRxiv</source>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maimela</surname>
<given-names>N. R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Fates of cd8&#x2b; t cells in tumor microenvironment</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>17</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2018.11.004</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maron</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Lozano-P&#xe9;rez</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>A framework for multiple-instance learning</article-title>. <source>Adv. neural Inf. Process. Syst.</source>, <fpage>570</fpage>&#x2013;<lpage>576</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohammadzadeh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lejeune</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Predicting mechanically driven full-field quantities of interest with deep learning-based metamodels</article-title>. <source>Extreme Mech. Lett.</source> <volume>50</volume>, <fpage>101566</fpage>. <pub-id pub-id-type="doi">10.1016/j.eml.2021.101566</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nielsen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lundegaard</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Worning</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lauem&#xf8;ller</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Lamberth</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Buus</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2003</year>). <article-title>Reliable prediction of t-cell epitopes using neural networks with novel sequence representations</article-title>. <source>Protein Sci.</source> <volume>12</volume>, <fpage>1007</fpage>&#x2013;<lpage>1017</lpage>. <pub-id pub-id-type="doi">10.1110/ps.0239403</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>O&#x2019;Donnell</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Rubinsteyn</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Laserson</surname>
<given-names>U.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Mhcflurry 2.0: improved pan-allele prediction of mhc class i-presented peptides by incorporating antigen processing</article-title>. <source>Cell. Syst.</source> <volume>11</volume>, <fpage>42</fpage>&#x2013;<lpage>48.e7</lpage>. <pub-id pub-id-type="doi">10.1016/j.cels.2020.06.010</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ong</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Pearce</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Computational design of sars-cov-2 spike glycoproteins to increase immunogenicity by t cell epitope engineering</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>19</volume>, <fpage>518</fpage>&#x2013;<lpage>529</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2020.12.039</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parker</surname>
<given-names>K. C.</given-names>
</name>
<name>
<surname>Bednarek</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Coligan</surname>
<given-names>J. E.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Scheme for ranking potential hla-a2 binding peptides based on independent binding of individual peptide side-chains</article-title>. <source>J. Immunol.</source> <volume>152</volume>, <fpage>163</fpage>&#x2013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.4049/jimmunol.152.1.163</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bhattacharya</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Canny</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Evaluating protein transfer learning with tape</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>32</volume>, <fpage>9689</fpage>&#x2013;<lpage>9701</lpage>. <pub-id pub-id-type="doi">10.1101/676825</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Verkuil</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Meier</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Canny</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Msa transformer</article-title>. <comment>bioRxiv</comment>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rao</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Meier</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sercu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ovchinnikov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rives</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Transformer protein language models are unsupervised structure learners</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/2020.12.15.422761</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reynisson</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Alvarez</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Paul</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Nielsen</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Netmhcpan-4.1 and netmhciipan-4.0: improved predictions of mhc antigen presentation by concurrent motif deconvolution and integration of ms mhc eluted ligand data</article-title>. <source>Nucleic acids Res.</source> <volume>48</volume>, <fpage>W449</fpage>&#x2013;<lpage>W454</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkaa379</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rives</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Goyal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Meier</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ott</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zitnick</surname>
<given-names>C. L.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Biological structure and function emerge from scaling unsupervised learning to 250 million protein sequences</article-title>. <comment>bioRxiv</comment>, <fpage>622803</fpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Pedrioli</surname>
<given-names>P. G.</given-names>
</name>
<name>
<surname>Wolski</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Scurtescu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Schmid</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Vizca&#xed;no</surname>
<given-names>J. A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>The systemhc atlas project</article-title>. <source>Nucleic acids Res.</source> <volume>46</volume>, <fpage>D1237</fpage>&#x2013;<lpage>D1247</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx664</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Teraguchi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Saputri</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Llamas-Covarrubias</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Davila</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Diez</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nazlica</surname>
<given-names>S. A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Methods for sequence and structural analysis of b and t cell receptor repertoires</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>18</volume>, <fpage>2000</fpage>&#x2013;<lpage>2011</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2020.07.008</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tong</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Blocks substitution matrix (blosum)</article-title>,&#x201d; in <source>Encyclopedia of systems biology</source> (<publisher-name>Springer</publisher-name>).</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Townsend</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Elliott</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cerundolo</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Foster</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Barber</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Tse</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Assembly of mhc class i molecules analyzed <italic>in vitro</italic>
</article-title>. <source>Cell.</source> <volume>62</volume>, <fpage>285</fpage>&#x2013;<lpage>295</lpage>. <pub-id pub-id-type="doi">10.1016/0092-8674(90)90366-m</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Attention is all you need</article-title>. <comment>arXiv preprint arXiv:1706.03762</comment>.</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veli&#x10d;kovi&#x107;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cucurull</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Casanova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Romero</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lio</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Graph attention networks</article-title>. <comment>arXiv preprint arXiv:1710.10903</comment>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vita</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Mahajan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Overton</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Dhanda</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Martini</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cantrell</surname>
<given-names>J. R.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>The immune epitope database (iedb): 2018 update</article-title>. <source>Nucleic acids Res.</source> <volume>47</volume>, <fpage>D339</fpage>&#x2013;<lpage>D343</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1006</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Voulodimos</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Doulamis</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Doulamis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Protopapadakis</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep learning for computer vision: a brief review</article-title>. <source>Comput. Intell. Neurosci.</source> <volume>2018</volume>, <fpage>1</fpage>, <lpage>13</lpage>. <pub-id pub-id-type="doi">10.1155/2018/7068349</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gifford</surname>
<given-names>D. K.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deepligand: accurate prediction of mhc class i ligands using peptide embedding</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>i278</fpage>&#x2013;<lpage>i283</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz330</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bickis</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>F.-X.</given-names>
</name>
<name>
<surname>Kusalik</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Optimally-connected hidden markov models for predicting mhc-binding peptides</article-title>. <source>J. Bioinforma. Comput. Biol.</source> <volume>4</volume>, <fpage>959</fpage>&#x2013;<lpage>980</lpage>. <pub-id pub-id-type="doi">10.1142/s0219720006002314</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.-M.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>M.-J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Graph neural networks and their current applications in bioinformatics</article-title>. <source>Front. Genet.</source> <volume>12</volume>, <fpage>690049</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2021.690049</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>