<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2024.1478201</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>LRMAHpan: a novel tool for multi-allelic HLA presentation prediction using Resnet-based and LSTM-based neural networks</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Mi</surname>
<given-names>Xue</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2813002"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Shaohao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ye</surname>
<given-names>Zheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dai</surname>
<given-names>Zhu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ding</surname>
<given-names>Bo</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1319470"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sun</surname>
<given-names>Bo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Shen</surname>
<given-names>Yang</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1380263"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Xiao</surname>
<given-names>Zhongdang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>State Key Laboratory of Bioelectronics, School of Biological Science and Medical Engineering, Southeast University</institution>, <addr-line>Nanjing</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Obstetrics and Gynecoloty, Zhongda Hospital, School of Medicine, Southeast University</institution>, <addr-line>Nanjing</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Jiangsu Sports Health Research Institute, Institute of Sports and Health</institution>, <addr-line>Nanjing</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Wouter Scheper, The Netherlands Cancer Institute (NKI), Netherlands</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Wentao Dai, Shanghai Institute for Biomedical and Pharmaceutical Technologies, China</p>
<p>Georgios Fotakis, Innsbruck Medical University, Austria</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Yang Shen, <email xlink:href="mailto:shenyang@seu.edu.cn">shenyang@seu.edu.cn</email>; Zhongdang Xiao, <email xlink:href="mailto:zdxiao@seu.edu.cn">zdxiao@seu.edu.cn</email>
</p>
</fn>
<fn fn-type="other" id="fn003">
<p>&#x2020;These authors share first authorship</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>28</day>
<month>11</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1478201</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>08</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>10</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Mi, Li, Ye, Dai, Ding, Sun, Shen and Xiao</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Mi, Li, Ye, Dai, Ding, Sun, Shen and Xiao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>The identification of peptides eluted from HLA complexes by mass spectrometry (MS) can provide critical data for deep learning models of antigen presentation prediction and promote neoantigen vaccine design. A major challenge remains in determining which HLA allele eluted peptides correspond to.</p>
</sec>
<sec>
<title>Methods</title>
<p>To address this, we present a tool for prediction of multiple allele (MA) presentation called LRMAHpan, which integrates LSTM network and ResNet_CA network for antigen processing and presentation prediction. We trained and tested the LRMAHpan BA (binding affinity) and the LRMAHpan AP (antigen processing) models using mass spectrometry data, subsequently combined them into the LRMAHpan PS (presentation score) model. Our approach is based on a novel pHLA encoding method that enables the integration of neoantigen prediction tasks into computer vision methods. This method aggregates MA data into a multichannel matrix and incorporates peptide sequences to efficiently capture binding signals.</p>
</sec> <sec>
<title>Results</title>
<p>LRMAHpan outperforms standard predictors such as NetMHCpan 4.1, MHCflurry 2.0, and TransPHLA in terms of positive predictive value (PPV) when applied to MA data. Additionally, it can accommodate peptides of variable lengths and predict HLA class I and II presentation. We also predicted neoantigens in a cohort of metastatic melanoma patients, identifying several shared neoantigens.</p>
</sec>
<sec>
<title>Discussion</title>
<p>Our results demonstrate that LRMAHpan significantly improves the accuracy of antigen presentation predictions.</p>
</sec>
</abstract>
<kwd-group>
<kwd>biomedical engineering</kwd>
<kwd>neoantigen prediction</kwd>
<kwd>deep learning</kwd>
<kwd>multi allelic HLA</kwd>
<kwd>MHC</kwd>
<kwd>antigen processing</kwd>
</kwd-group>
<counts>
<fig-count count="4"/>
<table-count count="3"/>
<equation-count count="9"/>
<ref-count count="49"/>
<page-count count="13"/>
<word-count count="6541"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Cancer Immunity and Immunotherapy</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>Peptide-HLA (pHLA) complexes consist of peptides that attach to human leukocyte antigens (HLA) and are presented to specialized immune cells, thereby initiating an immune response. HLA molecules are crucial for this process, as they present antigenic peptides on the cell surface for recognition by T cells (<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B2">2</xref>). This antigen presentation allows T cells to identify and attack infected or mutated cells. Infections can act as etiological factors in the development of various cancers. HLA molecules are integral to the anti-cancer immune response, playing key roles in the management of multiple cancer types, including lung, prostate, breast, and colon cancer (<xref ref-type="bibr" rid="B3">3</xref>&#x2013;<xref ref-type="bibr" rid="B8">8</xref>).</p>
<p>HLA-I genes are highly polymorphic, with HLA heavy chains encoded by three genes: HLA-A, HLA-B, and HLA-C. All three genes are polymorphic, constituting the most distinctive feature of HLA molecules, which leads to variability in peptide presentation (typically 8-11 amino acids) among individuals (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B10">10</xref>). Additionally, HLA-II molecules, located on human cells and consisting of three loci on chromosome 6 (DR, DQ and DP), are involved in the presentation of exogenous antigen (usually 13-25 amino acids) (<xref ref-type="bibr" rid="B11">11</xref>). The binding of peptides to HLA is the most critical and selective step in antigen presentation (<xref ref-type="bibr" rid="B12">12</xref>), making the identification of pHLA essential for developing effective immunotherapeutic cancer vaccines and studying infectious disease (<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B14">14</xref>). This highlights the need for in silico algorithms capable of accurately predicting pHLA molecules.</p>
<p>Several tools have been developed to address the challenges of neoantigen prediction, employing two main types of computational methods: single allele (SA) and multiple allele (MA) predictors. Both types typically consist of two predictive models: HLA-I binding affinity (BA) (<xref ref-type="bibr" rid="B15">15</xref>&#x2013;<xref ref-type="bibr" rid="B18">18</xref>) and antigen processing (AP) (<xref ref-type="bibr" rid="B19">19</xref>&#x2013;<xref ref-type="bibr" rid="B21">21</xref>) predictors. Recent advancements in mass spectrometry (MS) technology have facilitated the identification of peptides in high-throughput experiments, creating opportunities for developing neoantigen predictors. MHCflurry 2.0 (<xref ref-type="bibr" rid="B22">22</xref>) has integrated AP and BA predictors to significantly enhance prediction accuracy. Traditionally, published models segment MA mass spectrometry (MS) sequences into SA MS sequences for independent integration of pHLA into predictive models. Conversely, our approach directly integrates MA and peptides into the model as a cohesive entity, enhancing prediction accuracy through interactions between MA and peptides. Furthermore, combining peptide sequences with MA predictors (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B23">23</xref>) offers greater intuitiveness and alignment with real human environments. However, studies utilizing multi-allelic (MA) data remain limited. Specifically, when considering the use of MA data as a whole input based on input patterns, the only available MA predictor is LRMAHpan.</p>
<p>ResNet (<xref ref-type="bibr" rid="B24">24</xref>) has been successfully applied in image recognition, yet the challenging of using ResNet for antigen presentation prediction has not been thoroughly explored. The shortcut connections of ResNet network significantly reduce the complexity of training deep neural networks (<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B26">26</xref>). The ResNet architecture consists of multiple similar residual blocks arranged in series. The Coordinate Attention (<xref ref-type="bibr" rid="B27">27</xref>) (CA) mechanism captures location and channel relationships, enabling the network to gather information from a larger area without significant resource consumption (<xref ref-type="bibr" rid="B28">28</xref>&#x2013;<xref ref-type="bibr" rid="B30">30</xref>).</p>
<p>In this study, we address the limitations of preprocessing that arise from the one-to-one correspondence between peptide sequences and HLA types by utilizing ResNet_CA-based deep convolutional neural networks for the BA model and LSTM neural network for the AP model. LRMAHpan introduces a novel coding approach that utilizes 6-channel pHLA encoding as input data for residual networks, with each channel representing one of the six HLA types. LRMAHpan is the first ResNet_CA-based method for predicting antigen presentation, leveraging data from multiple allele (MA) mass spectrometry (MS) datasets to achieve accurate predictions. By incorporating a CA module, LRMAHpan effectively captures crucial binding signals directly from MA MS raw data, thus improving binding accuracy across various alleles and peptide sequences. The model can handle peptide sequences of variable lengths (8-11 amino acids), and we also trained and validated its performance in pHLA-II presentation by adjusting the number of channels and the length of peptide sequences (13-25 amino acids). Finally, we assembled different AP and BA predictors to forecast the potential of MA HLA in presenting sequences, resulting in the development of the presentation score (PS) (LRMAHpan PS). Our findings indicate that PS predictor outperforms both AP and BA models, demonstrating superior performance compared to commonly used tools such as NetMHCpan 4.1 (<xref ref-type="bibr" rid="B31">31</xref>), MHCflurry 2.0 and TransPHLA (<xref ref-type="bibr" rid="B32">32</xref>).</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="s2_1">
<title>Datasets</title>
<p>We used the multiple allele (MA) mass spectrometry (MS) datasets curated by EDGE (<xref ref-type="bibr" rid="B23">23</xref>), integrating them with an additional dataset derived from MHCflurry2.0 (<xref ref-type="bibr" rid="B22">22</xref>) to train the final version of our model. Negative samples were generated from peptides sourced from the reference proteome (SwissProt) that were not detected by mass spectrometry in the original samples. Specifically, we randomly sampled two segments from each negative peptide sequence, with the length of each segment reflecting the distribution of lengths in the positive dataset. From each sample, we randomly selected 1,800 data points, ensuring that no peptide sequences overlapped with those present in the positive dataset. Consequently, the final dataset maintained a 1:4 ratio of positive to negative samples, with a training set comprising 221,061 positive samples. HLA typing for the MA in the training set is detailed in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>. Due to the frequent sharing of high-frequency alleles among patients, our analysis revealed a total of 118 unique HLA typing combinations, each associated with the presentation of more than 30 peptide sequences.</p>
<p>To mitigate variability associated with data preprocessing, we utilized existing post-processed training datasets to directly assess prediction systems (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>). The test dataset was obtained from MULTIALLELIC-RECENT benchmark dataset of MixMHCpred 2.0.2 (<xref ref-type="bibr" rid="B33">33</xref>), which includes mass spectrometry (MS) data from tumor samples of ten patients. The ratio of presenting peptides to non-presenting peptides in this dataset is 1:99. As predicted events (i.e., presenting peptides) are rare, achieving a high positive predictive value (PPV) becomes increasingly challenging, resulting in a more stringent evaluation of the model&#x2019;s performance. To mitigate the impact of negative sample selection on the results, we also employed multi-allelic dataset provided by the IEDB database, which includes both presenting and non-presenting peptides, maintaining a 1:1 ratio for predictions (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Note S2</bold>
</xref>). Importantly, there is no overlap between the test and training datasets. In the training set, we excluded data from patients with incomplete HLA typing to enable the model to learn more accurate features of multi-allelic types. Consequently, the model prefers complete HLA data during predictions. If HLA typing information for a patient is incomplete at the time of prediction, our model can still process the data by supplementing it with the patient&#x2019;s known typing information. Detailed usage instructions are available on GitHub and in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Note S2</bold>
</xref>.</p>
<p>Data from cBioportal (<xref ref-type="bibr" rid="B34">34</xref>, <xref ref-type="bibr" rid="B35">35</xref>) were retrieved from metastatic melanoma cohorts to predict neoantigens using LRMAHpan. The cohort was constructed by sequencing the whole exomes of 38 pairs of pre-treatment melanoma tumors and normal tissues. This data includes mutation maps in MAF format for 38 cases and RPKM expression data obtained from mRNA analysis for 27 cases. Additionally, the dataset includes results of HLA class I and class II typing.</p>
</sec>
<sec id="s2_2">
<title>HLA representation</title>
<p>HLA typing was carried out using the OptiType 1.3.1 HLA analysis software packages. This tool was utilized to generate HLA types from matched normal DNA samples, allowing for accurate computational HLA typing. HLA class I alleles are represented by a &#x201c;pseudo sequence&#x201d; proposed by NetMHCpan (<xref ref-type="bibr" rid="B36">36</xref>). In our approach, we utilize the pseudo sequence generated by MHCflurry 2.0, which differs from the NetMHCpan pseudo sequence in that it has a length of 37. In addition to the 34 peptide contact positions contained in the NetMHCpan pseudo sequence, we incorporate three new positions (115, 126, and 23). These additional positions are selected to differentiate alleles that share the same NetMHCpan pseudo sequence. The pseudo-sequence of HLA class II is derived from the representation offered by NetMHCIIpan3.0 (<xref ref-type="bibr" rid="B37">37</xref>), which includes amino acid residues critical for peptide binding. It comprises 15 residues from the &#x3b1; chain and 19 residues from the &#x3b2; chain of HLA class II molecules, resulting in specific residues at defined positions. For the &#x3b1; chain, these positions are 9, 11, 22, 24, 31, 52, 53, 58, 59, 61, 65, 66, 68, 72, and 73. For the &#x3b2; chain, the positions are 9, 11, 13, 26, 28, 30, 47, 57, 67, 70, 71, 74, 77, 78, 81, 85, 86, 89, and 90. Consequently, the final length of the pseudo-sequence for HLA class II molecules totals 34 residues (15 from the &#x3b1; chain and 19 from the &#x3b2; chain).</p>
</sec>
<sec id="s2_3">
<title>Peptide-HLA encoding</title>
<p>The input to the LRMAHpan BA network was generated by scanning six HLA allele pseudosequences and peptide sequences (see <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>). For each peptide sequence, a 3-dimensional feature matrix M (size 6&#xd7;59&#xd7;22) was constructed, comprising six channels of size 59&#xd7;22, with each channel corresponding to one HLA allele type. This design aims to capture the signal from HLA peptide binding. Here, 59 represents the sum of the corresponding peptide length, pseudosequence length, and reverse peptide length, with all peptides padded to a maximum length of 59 using padding characters. Furthermore, 22 represents 20 common amino acids sequences, along with the padding marker &lt;PAD&gt;. Each amino acid in the peptide sequence was vectorized using a one-hot encoding scheme (20 common amino acids + &lt;PAD&gt;). Consequently, each allele peptide is represented by a two-dimensional vector of size (59, 22).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>The structure of LRMAHpan PS predictor, LRMAHpan BA predictor and LRMAHpan AP predictor. <bold>(A)</bold> BA model input representations, for example, AYTSGLEY coding+ HLA pseudosequence coding+ YELGSTYA coding. Notably, each BA model input consists of six such data representations. <bold>(B)</bold> The input scheme accommodates peptides with variable lengths, capable of handling peptides of any length by selecting the maximum length, set here at 11. <bold>(C)</bold> The major sub-module (CA module) of BA predictor. <bold>(D)</bold> The LRMAHpan BA predictor adopts a Resnet structure. <bold>(E)</bold> The LRMAHpan AP predictor employs an LSTM structure. <bold>(F)</bold> The LRMAHpan PS predictor is proposed as a composite of two models <bold>(D, E).</bold> The LRMAHpan PS model is designed to predict neoantigens, combining multiple AP and BA models through the calculation of mean values.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1478201-g001.tif"/>
</fig>
<p>The peptide lengths range from 8 to 11 amino acids (AA), as this range encompasses about ninety-five percent of HLA class I presented peptides. The LSTM model employed the nn.Embedding module from PyTorch, which initializes embedding weights randomly. In the LRMAHpan AP mode, peptide sequences were vectorized using a parameterized embedding method, and peptides of multiple lengths (8-11 AA) were represented as vectors of fixed length by adding amino acid alphabets with padding characters and ensuring that all peptides were filled to a maximum length of 11 (see <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>). For the HLA class II data, peptides with lengths ranging from 13 to 25 amino acids were included.</p>
</sec>
<sec id="s2_4">
<title>The construction of neural networks</title>
<p>The LRMAHpan BA model incorporates a Coordinate Attention (CA) module into the ResNet residual block module (see <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1C</bold>
</xref>). It accepts any intermediate feature tensor <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> as input and outputs a transformed tensor with augmented representations <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> of the same size as <inline-formula>
<mml:math display="inline" id="im3">
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula>. To balance data volume and model size, we utilize a ResNet18 model comprising 17 convolutional layers and one fully connected layer, structured as follows (see <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1D</bold>
</xref>). The input matrix is processed through the initial convolutional layer with a kernel size of 3&#xd7;3, followed by four series of Residual Blocks, and then passed through AdaptiveAvgPool2d. The output of the final block is fed into a fully connected layer with an output size of 2, predicting whether the peptide can be presented by HLA.</p>
<p>The LRMAHpan BA model adds a CA module after the BatchNorm layer in the residual block module to enhance the feature representation. This CA approach addresses the challenge of location information loss from 2D global pooling by partitioning channel attentions into two parallel 1D signature encodings, effectively integrating spatial coordinate information into resultant attention maps. Notably, the CA technique features adaptability and a lightweight design, leveraging collected location data for precise region-of-interest capture and effectively capturing inter-channel relationships.</p>
<p>In channel attention mechanisms, global pooling is typically employed to comprehensively encode spatial information. However, this approach compresses global spatial data into a channel descriptor, which poses challenges in retaining positional information. To facilitate attention blocks in capturing distant spatial interactions with precise positional details, Coordinate Attention (CA) Blocks decompose global pooling into a pair of 1D feature encoding operations, as shown in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>. Given input X, we utilize two spatial extents of pooling kernels (H, 1) or (1, W), to encode each channel along the horizontal and vertical coordinates, respectively. The squeeze step for the c-th channel can be expressed as follows:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>H</mml:mi>
</mml:munderover>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>W</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Thus, the output of the c-th channel at height h can be formulated as:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>h</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>W</mml:mi>
</mml:mfrac>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&lt;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Similarly, the output of the c-th channel at width w can be expressed as:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>w</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mtext>H</mml:mtext>
</mml:mfrac>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&lt;</mml:mo>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Here, <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the output associated with the c-th channel. The input X is derived directly from a convolutional layer with a fixed kernel size, representing a set of local descriptors. The squeeze operation facilitates the aggregation of global information.</p>
<p>Upon obtaining the aggregated feature maps generated by <xref ref-type="disp-formula" rid="eq2">Equations 2</xref>, <xref ref-type="disp-formula" rid="eq3">3</xref>, we concatenate them and pass them through a shared 1 &#xd7; 1 convolutional transformation function F1, yielding:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where [&#xb7;, &#xb7;] denotes the concatenation operation along the spatial dimension, &#x3b4; is a non-linear activation function, and <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the intermediate feature map that encodes spatial information in both the horizontal and vertical directions. We then split <inline-formula>
<mml:math display="inline" id="im6">
<mml:mi>f</mml:mi>
</mml:math>
</inline-formula> along the spatial dimension into two separate tensors <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Two additional 1 &#xd7; 1 convolutional transformations, <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, are utilized to separately transform <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to tensors with the same channel number as the input X, yielding.</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msup>
<mml:mi> &#x261;</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>h</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msup>
<mml:mi>&#x261;</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>w</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Here, &#x3c3; is the sigmoid function, and the output of our coordinate attention block Y can be written as:</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:msubsup>
<mml:mi>g</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>h</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:msubsup>
<mml:mi>g</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>w</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>To enhance the robustness and generalization capabilities of the BA model, our approach combines Sharpness-Aware Minimization (SAM) (<xref ref-type="bibr" rid="B38">38</xref>) and SGD (<xref ref-type="bibr" rid="B39">39</xref>) to achieve a balance between training duration and generalization capacity. Regardless of the gradient descent or optimization approach, the goal of training the model is to identify the parameters that minimize loss value. Notably, in contrast to other optimization methods, SAM achieves superior generalization by enhancing the training process through the simultaneous minimization of both loss value and loss sharpness. Furthermore, it explores parameters exclusively within neighborhoods exhibiting consistently low loss values, resulting in a flatter loss hyperplane compared to alternative optimization methods, thereby augmenting the model&#x2019;s generalization capabilities. However, SAM requires double the training time due to computing the sharpness-aware gradient twice.</p>
<p>Based on the characteristics of the presented peptides, we propose a novel antigen peptide processing predictor based on the Bi-LSTM (<xref ref-type="bibr" rid="B40">40</xref>) framework, corresponding to LRMAHpan AP. The LRMAHpan AP predictor (see <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1E</bold>
</xref>) comprises several key layers: embedding, spatial dropout, LSTM, GRU, Relu (<xref ref-type="bibr" rid="B41">41</xref>), maximum pooling, average pooling, and fully connected layers. SpatialDropout (<xref ref-type="bibr" rid="B42">42</xref>) randomly eliminates several feature dimensions. We utilize embedding-encoded peptide representations as the input to our model. Notably, the embedding dimension within the neural network is set at 100, while the hidden layers of both LSTM and GRU consist of 128 neurons each. Additionally, the largest pooling layer is connected to the average pooling layer to facilitate feature reuse, enhancing training efficiency and serving as input for subsequent layers.</p>
<p>This work introduces LRMAHpan PS as the ultimate presentation model, achieved by averaging the outcomes of LRMAHpan BA and LRMAHpan AP (see <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1F</bold>
</xref>).</p>
</sec>
<sec id="s2_5">
<title>Model training</title>
<p>For model training, we divided dataset (refer to the Dataets section) into multiple subsets: 95% for training and 5% for validation, utilizing different random seeds. A larger training dataset enables the model to learn more effectively and capture diverse binding patterns, which are crucial for its performance. The remaining 5% of the data is utilized as a validation set to evaluate the model&#x2019;s performance and ensure its ability to generalize to unseen data. This approach aimed to identify the hyperparameters that minimize the loss value of the LRMAHpan BA model. We employed early stopping to monitor the performance metric, halting training when the performance on the validation set began to deteriorate. The neural network was trained using the SGD optimizer with a cross-entropy loss function. Training was conducted with a batch size of 128, an initial learning rate of 0.1 and a momentum value of 0.9. The learning rate was subsequently reduced to 0.02, 0.004, and 0.0008 at the 60th, 120th, and 180th iterations, respectively. The total training process encompassed 200 iterations. For optimizing the LRMAHpan AP model, we applied the same strategy. In this case, we divided the peptides into a training set (90%) and a validation set (10%), keeping all other training parameters consistent with those used for the LRMAHpan BA model.</p>
</sec>
<sec id="s2_6">
<title>Model selection</title>
<p>The imbalance between positive and negative samples poses a significant challenge in tumor neoantigen prediction, potentially biasing model predictions towards the majority class. To address this issue, we employed an effective technique known as EasyEnsemble (<xref ref-type="bibr" rid="B43">43</xref>). This technique integrates undersampling and demonstrates strong performance in real-world scenarios. We set the ratio of positive to negative samples at 1:4, training the model with the sampled negative samples and all positive samples. The F1 score of the validation set was used as the performance metric for each model. Subsequently, we selected several top-performing models for ensemble averaging. The ensemble for LRMAHpan BA comprised nine models, while the ensemble for LRMAHpan AP included six models. During testing, the final prediction was generated by averaging the output probabilities from these selected models.</p>
</sec>
<sec id="s2_7">
<title>Quantitative and statistical indicators</title>
<p>The model primarily employed PPV as the performance metrics, defined as PPV=NTP/(NTP+NFP), where NTP represents the number of true positives and NFP represents the number of false positives. The performance evaluation utilized Average Precision (AP) to assess the average precision and recall of a classification model at various thresholds. AP is particularly suitable for imbalanced datasets as it emphasizes the model&#x2019;s ability to identify positive samples. For continuous PR curves, the formula for AP is given by:</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mn>1</mml:mn>
</mml:munderover>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>For discrete PR curves, the formula for AP is expressed as:</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mtext>AP</mml:mtext>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s2_8">
<title>Contrast with currently available tools</title>
<p>The purpose of this article is to evaluate LRMAHpan BA and LRMAHpan AP against the most advanced binding affinity predictor and presentation predictor (NetMHCPan 4.1, MHCflurry 2.0, TransPHLA). Our approach for assessing single allele (SA) predictors (NetMHCpan 4.1, MHCflurry 2.0, TransPHLA) using multiple allele (MA) data involves combining peptide sequences with each HLA typing separately, in accordance with the input characteristics of the predictors (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Note S1</bold>
</xref>). This method yields optimal results compared to our model (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S3</bold>
</xref>).</p>
<p>The benchmark proposed by MHCflurry 2.0 was employed for performance comparison. To ensure a fair evaluation, the training dataset was omitted from the KESKIN MA dataset, as the MHCflurry 2.0 BA training process utilized a KESKIN SA cell line, which could provide an advantage in the MA dataset. Additionally, the exclusion of these datasets from the benchmark was motivated by the presence of the MULTIALLELIC-OLD data within the LRMAHpan training dataset. The final dataset used for comparison was a set of 10 datasets known as MULTIALLELIC_B, which contained a total of 18,472 presented peptides.</p>
<p>In the performance comparison, IC50 values were transformed into probability values ranging from 0 to 1. This facilitated comparisons between LRMAHpan Presentation Score (PS) and MHCflurry 2.0 PS, as well as between LRMAHpan BA and both MHCflurry 2.0 BA and NetMHCPan 4.1 BA. When comparing LRMAHpan AP with MHCflurry AP, it is important to highlight that MHCflurry 2.0 utilized a final training set comprising 493,473 MS data points and 219,596 affinity measurements, while LRMAHpan relied on only 221,061 presented MS data points. This indicates that our model can extract accurate features and make precise predictions using limited data.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s3_1">
<title>Prediction performance of LRMAHpan BA</title>
<p>To evaluate the performance of the LRMAHpan BA predictor based on the ResNet_CA network, we screened ten samples from MULTIALLELIC benchmark (see Methods for more details), ensuring the inclusion of six HLA alleles as an independent test dataset (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). Each peptide sequence was combined with an HLA pseudosequence and a reverse peptide sequence, then encoded into a vector (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>). Additionally, each peptide sequence could be separately combined with six HLA typings to form a six-channel data input for training and testing. A benchmark was established using public datasets of HLA ligands identified by mass spectrometry (MS) (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>). We compared the performance of our model to that of the current state-of-the-art methods, MHCflurry2.0 BA and NetMHCpan4.1 BA (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2A</bold>
</xref>), which are widely used for predicting HLA ligands. LRMAHpan BA demonstrated superior performance compared to both MHCflurry2.0 BA and NetMHCpan4.1 BA when applied to test data. The positive predictive value (PPV) was calculated at the recall rate was 50% on a test set composed of ten subsets with a 1:99 ratio of positive to negative samples. For instance, the PPV of LRMAHpan BA, MHCflurry2.0 BA, and Netmhcpan4.1 BA were 0.477, 0.151 and 0.080, respectively (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1A</bold>
</xref>). In the dataset 29_14-TISSUE, which contained the largest number of positive samples, the PPV of LRMAHpan BA was 8.9 times higher than that of MHCflurry2.0 BA and 13.9 times higher than that of NetMHCpan4.1 BA (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S1B, G</bold>
</xref>). To assess whether this advantage was consistent across different datasets, we tested data with positive to negative ratios of 1:1 and 1:9. The results indicated that regardless of the ratio, our PPV values were superior to those of existing tools (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1F</bold>
</xref>). Similarly, in dataset 637-13-TISSUE, the PPV of LRMAHpan BA was 4.1 times higher than that of MHCflurry2.0 BA and 7.9 times higher than that of NetMHCpan4.1 BA (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S1C, G</bold>
</xref>). The excellent performance of the BA model in terms of PPV may be attributed to the multi-allelic model&#x2019;s ability to recalled fewer false positive prediction predictions under the same datasets. Despite undergoing identical validation procedures, our model is unique in simultaneously considering six alleles, whereas the SA predictor necessitates multiple iterations involving peptide sequences and HLA typing six times. This distinction may result in higher recall rates for SA predictors, along with an increased likelihood of false positives.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Independent test sets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Sample id</th>
<th valign="top" align="center">#Pos</th>
<th valign="top" align="center">#Neg</th>
<th valign="top" align="center">HLA</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">11-002-S1-TISSUE</td>
<td valign="top" align="center">946</td>
<td valign="top" align="center">93654</td>
<td valign="top" align="center">A0301 A2402 B3503 B4402 C1203 C1203</td>
</tr>
<tr>
<td valign="top" align="center">10-002-S1-TISSUE</td>
<td valign="top" align="center">431</td>
<td valign="top" align="center">42669</td>
<td valign="top" align="center">A0201 A3101 B1302 B5801 C0602 C0701</td>
</tr>
<tr>
<td valign="top" align="center">BCN-018-TISSUE</td>
<td valign="top" align="center">935</td>
<td valign="top" align="center">92565</td>
<td valign="top" align="center">A0201 A2901 B0702 B2705 C0102 C1505</td>
</tr>
<tr>
<td valign="top" align="center">CPH-09-TISSUE</td>
<td valign="top" align="center">1527</td>
<td valign="top" align="center">151173</td>
<td valign="top" align="center">A0201 A3201 B2705 B4402 C0501 C0202</td>
</tr>
<tr>
<td valign="top" align="center">CPH-07-TISSUE</td>
<td valign="top" align="center">1816</td>
<td valign="top" align="center">179784</td>
<td valign="top" align="center">A0201 A0201 B3501 B2705 C0202 C0401</td>
</tr>
<tr>
<td valign="top" align="center">29-14-TISSUE</td>
<td valign="top" align="center">4049</td>
<td valign="top" align="center">400851</td>
<td valign="top" align="center">A0201 A3201 B4001 B1302 C0304 C0602</td>
</tr>
<tr>
<td valign="top" align="center">637-13-TISSUE</td>
<td valign="top" align="center">2386</td>
<td valign="top" align="center">236214</td>
<td valign="top" align="center">A0101 A2402 B5101 B0801 C0701 C0102</td>
</tr>
<tr>
<td valign="top" align="center">LEIDEN-005-TISSUE</td>
<td valign="top" align="center">2066</td>
<td valign="top" align="center">204534</td>
<td valign="top" align="center">A0201 A2501 B3501 B1801 C1203 C0401</td>
</tr>
<tr>
<td valign="top" align="center">CPH-08-TISSUE</td>
<td valign="top" align="center">3008</td>
<td valign="top" align="center">297792</td>
<td valign="top" align="center">A3201 A2601 B3801 B4002 C0202 C1203</td>
</tr>
<tr>
<td valign="top" align="center">LEIDEN-004-TISSUE</td>
<td valign="top" align="center">1308</td>
<td valign="top" align="center">129492</td>
<td valign="top" align="center">A0301 A0201 B0702 B0702 C1203 C0702</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Benchmarking the performance of the models. <bold>(A)</bold> Performance comparison of PPV of the BA models against other predictors, with each point representing a single experiment. <bold>(B)</bold> AUC values of LRMAHpan. <bold>(C, D)</bold> PPV of LRMAHpan PS is contrasted with other predictors. <bold>(E)</bold> Violin plot display PPV and AUC values of LRMAHpan AP alongside MHCflurry2.0 AP across ten independent test sets. <bold>(F)</bold> The structural depiction of the experimental complex involving the epitope IMDQVPFSV presented by HLA-A:02*01, showcasing detailed residue interactions. The &#x3b1; and &#x3b2; chains within the HLA-A:02*01 structure are highlighted in green and orange, respectively, while non-covalent interactions between HLA and peptide residues are illustrated by dashed lines using PyMOL. <bold>(G)</bold> AUC values of 9 models on 10 independent test sets. The symbol ** indicates that the p-value is less than 0.01.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1478201-g002.tif"/>
</fig>
<p>In addition, we trained a series of models and found that using ResNet with CA resulted in slightly higher performance compared to models without CA (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>), demonstrating an average improvement of 0.02 in the area under the curve (AUC). The PPV and AUC values of LRMAHpan BA across the ten test sets were consistently higher than those of MHCflurry2.0 BA and NetMHCPan4.1 BA (see <xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2A, G</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S5A, B</bold>
</xref>). These observations illustrate that LRMAHpan BA possesses powerful feature extraction capabilities, generalizability and advantages in large datasets. Overall, LRMAHpan BA significantly improved predictive performance for MA presentation.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Compare the AUC with and without CA module in the test sets of the BA predictor.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Sample id</th>
<th valign="top" align="center">ResNet18</th>
<th valign="top" align="center">ResNet18_CA</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">CPH-09-TISSUE</td>
<td valign="top" align="center">0.90</td>
<td valign="bottom" align="center">0.92 (+0.02)</td>
</tr>
<tr>
<td valign="top" align="center">CPH-07-TISSUE</td>
<td valign="top" align="center">0.91</td>
<td valign="bottom" align="center">0.93 (+0.02)</td>
</tr>
<tr>
<td valign="top" align="center">CPH-08-TISSUE</td>
<td valign="top" align="center">0.93</td>
<td valign="bottom" align="center">0.95 (+0.02)</td>
</tr>
<tr>
<td valign="top" align="center">10-002-S1-TISSUE</td>
<td valign="top" align="center">0.95</td>
<td valign="bottom" align="center">0.97 (+0.02)</td>
</tr>
<tr>
<td valign="top" align="center">LEIDEN-005-TISSUE</td>
<td valign="top" align="center">0.92</td>
<td valign="bottom" align="center">0.95 (+0.03)</td>
</tr>
<tr>
<td valign="top" align="center">11-002-S1-TISSUE</td>
<td valign="top" align="center">0.93</td>
<td valign="bottom" align="center">0.95 (+0.02)</td>
</tr>
<tr>
<td valign="top" align="center">LEIDEN-004-TISSUE</td>
<td valign="top" align="center">0.89</td>
<td valign="bottom" align="center">0.91 (+0.02)</td>
</tr>
<tr>
<td valign="top" align="center">BCN-018-TISSUE</td>
<td valign="top" align="center">0.86</td>
<td valign="bottom" align="center">0.88 (+0.02)</td>
</tr>
<tr>
<td valign="top" align="center">637_13-TISSUE</td>
<td valign="top" align="center">0.90</td>
<td valign="bottom" align="center">0.91 (+0.01)</td>
</tr>
<tr>
<td valign="top" align="center">29_14-TISSUE</td>
<td valign="top" align="center">0.95</td>
<td valign="bottom" align="center">0.96 (+0.01)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_2">
<title>Prediction performance of LRMAHpan AP</title>
<p>Comparing the predictive capabilities of LRMAHpan AP and LRMAHpan BA reveals some interesting insights. The LRMAHpan AP predictor outperforms MHCflurry 2.0 AP predictor in terms of PPV and AUC indicators (see <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2E</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S5C</bold>
</xref>). To investigate whether the AP predictor differs from the BA predictor in feature extraction, we evaluated LRMAHpan AP and LRMAHpan BA models using ten test sets from the MLTIALLELIC_B dataset (see <xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2A, B</bold>
</xref>). LRMAHpan AP model solely utilizes mass spectrometry-derived peptide sequences as input (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S3</bold>
</xref>), while LRMAHpan BA model integrates data from six HLA types along with peptide sequences (see <xref ref-type="fig" rid="f1">
<bold>Figures&#xa0;1A, D</bold>
</xref>).</p>
<p>Interestingly, nine out of ten LRMAHpan BA samples exhibited higher PPV compared to LRMAHpan AP (see <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2A</bold>
</xref>). Regarding AUC values, the LRMAHpan AP outperformed the LRMAHpan BA in six out of ten samples (see <xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2B</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S5D</bold>
</xref>). The mean AUC of LRMAHpan AP predictor reached 0.92 (0.86-0.96), suggesting that the AP model effectively captures meaningful signals. Further comparisons of Recall, Accuracy, and F1 value on independent test sets highlight performance differences between the LRMAHpan AP and the LRMAHpan BA models (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2</bold>
</xref>). The LRMAHpan BA demonstrates higher accuracy and F1 score, while the LRMAHpan AP shows a higher recall rate.</p>
<p>Overall, LRMAHpan BA outperforms LRMAHpan AP in terms of predictive performance, partially attributed to the use of MA data and an improved approach for encoding peptide sequences. This enhanced performance can be attributed not only to the features of the training dataset (HLA-presented peptides) but also to the overall model design. The new model framework allows for learning connections between multiple alleles, rather than being limited to a single allele. In contrast, our AP model, which combines LSTM and GRU layers, exhibits a slight improvement in performance (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S7</bold>
</xref>).</p>
</sec>
<sec id="s3_3">
<title>Prediction performance of LRMAHpan PS</title>
<p>Furthermore, we explored whether the combination of the LRMAHpan AP and the LRMAHpan BA predictors could achieve superior prediction results. We subsequently compared the LRMAHpan PS model with several others, including MHCflurry 2.0 AP, MHCflurry 2.0 BA, MHCflurry 2.0 PS, NetMHCpan 4.1 EL, NetMHCpan 4.1 BA, and TransPHLA (see <xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2C, D</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figures S5E, F</bold>
</xref>). The LRMAHpan PS exhibits an average PPV higher than those of MHCflurry 2.0 PS, NetMHCpan 4.1 EL, and TransPHLA, with values of 0.4747, 0.2615, 0.1534, and 0.0642, respectively (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1D</bold>
</xref>). Across all samples, LRMAHpan PS shows higher AUC values compared to MHCflurry 2.0 PS, NetMHCpan 4.1 EL, and TransPHLA, achieving values of 0.9329, 0.8947, 0.8518, and 0.8157, respectively (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1E</bold>
</xref>).</p>
<p>In comparison with MHCnuggets (<xref ref-type="bibr" rid="B44">44</xref>) and MixMHCPred, LRMAHpan exhibited superior performance across both AUC and AP metrics, as illustrated in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S6</bold>
</xref>. We tested our model on well-studied HLA-peptide samples, such as the IMDQVPFSV epitope presented by HLA-A*02:01, demonstrating that LRMAHpan accurately predicts the potential presentation of this peptide by the patient&#x2019;s HLA (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S4A</bold>
</xref>). Correlation analysis with HLA typing (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S4B</bold>
</xref>) using the PSSM matrix (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S4C</bold>
</xref>) indicates that IMDQVPFSV can be presented by either HLA-A*02:01 or HLA-C*0501. Experimental data further confirm the binding of IMDQVPFSV and HLA-A*02:01, providing additional evidence for the reliability of our model (see <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2F</bold>
</xref>).</p>
<p>To assess the robustness of our model, we obtained mass spectrometry data for an ovarian cancer patient from Dao (<xref ref-type="bibr" rid="B45">45</xref>), encompassing a total of 1,874 presentation instances. The HLA typing included HLA-A*02:01/A*01:01, HLA-B*57:01/B*07:05, and HLA-C*06:02/C*15:05. We evaluated LRMAHpan PS and NetMHCpan 4.1 using various performance metrics&#x2014;AUC, Recall, Precision, F1, ACC, AP, and Matthews correlation coefficient (MCC) &#x2014;across different positive-to-negative sample ratios (1:1, 1:10, and 1:100), as illustrated in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>. These metrics serve distinct purposes, with AUC and AP values providing threshold-independent evaluations. NetMHCpan 4.1 performs admirably at a positive-to-negative sample ratio of 1:1 but exhibits a significant drop in accuracy as the proportion of negative samples increases. In contrast, LRMAHpan PS demonstrates commendable performance in terms of precision, F1, ACC, AP, and MCC. Notably, at a positive-to-negative sample ratio of 1:100, LRMAHpan PS is poised to predict more authentic neoantigens due to its higher precision and AP values. Given the inherent imbalance between presented and non-presented antigens in real human settings, with non-presented antigens typically outnumbering presented ones, the performance metrics at a ratio of 1:100 are more reflective of real-world scenarios. This underscores the robustness and fidelity of our model predictions.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Generalization and robustness validation results. <bold>(A)</bold> Performance of PS model in predicting pHLA-II. <bold>(B)</bold> In K562 cell lines, Comparison of AUC, Recall, Precision, F1, ACC, AP and MCC values of LRMAHpan PS and NetMHCpan4.1 with positive and negative sample ratios of 1:1, 1:10 and 1:100, respectively.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1478201-g003.tif"/>
</fig>
</sec>
<sec id="s3_4">
<title>Class II model proof of concept</title>
<p>We evaluated whether the prediction model we proposed can also be applied to class II HLA peptide presentation. We utilized class II mass spectrometry data from the MARIA (<xref ref-type="bibr" rid="B46">46</xref>) dataset, where each peptide corresponds to two HLA class II alleles, both expressing HLA-DRB1. The preprocessing steps included data deduplication, after which the dataset was divided into training and validation subsets. The AUC and AP values of the validation set were used as evaluation criteria. The model architecture and training methodology were consistent with those employed for predicting HLA-I peptide presentation, with the notable exception of incorporating two channels. Next, we evaluated the performance of LRMAHpan PS against the K562 DRB1*01:01 benchmark dataset from MARIA. which comprised 1,361 positive and 1,361 negative samples. We plotted the ROC and PR curves for the MARIA, NetMHCIIpan 4.0, and LRMAHpan PS, calculating their respective AUC and AP values. The results were 0.885 and 0.879 for MARIA, 0.765 and 0.824 for NetMHCIIpan 4.0, and 0.875 and 0.864 for LRMAHpan PS, respectively (see <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>). Comparative analysis reveals that the AUC and AP values of MARIA and LRMAHpan PS exceed those of NetMHCIIpan 4.0, with LRMAHpan PS demonstrating comparable efficacy to MARIA. These findings underscore the robust generalization and migratory capabilities of our model framework.</p>
</sec>
<sec id="s3_5">
<title>Examples of neoantigen prediction in metastatic melanoma cohorts</title>
<p>We utilized Maftools (<xref ref-type="bibr" rid="B47">47</xref>) to visualize the cohort and assess the mutation status of all metastatic melanoma samples. The primary categorization of variations included missense mutation, with single nucleotide polymorphisms (SNPs) being the predominant variation type, characterized notably by the frequent occurrence of C &gt; T transitions. Each sample exhibited significant variability in mutation burden, with a median of 497 mutations. TTN (84%) and MUC16 (78%) emerged as genes with substantial mutational frequencies (see <xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4A&#x2013;F</bold>
</xref>). The waterfall plot demonstrates that some genes were altered multiple times across different samples (see <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4G</bold>
</xref>). Comparing the mutation burden of metastatic melanoma to 33 other cancers in the TCGA revealed a notably high mutation load in melanoma (see <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4H</bold>
</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Mutational landscape of a metastatic melanoma cohort. <bold>(A)</bold> Overall variant classification by cohort. <bold>(B)</bold> Overall variant type by cohort. <bold>(C)</bold> Type of single nucleotide variation. <bold>(D)</bold> Number of variants per sample. <bold>(E)</bold> Cohort variant classification profile. <bold>(F)</bold> Top ten genes with the largest number of mutations. <bold>(G)</bold> Mutant landscape waterfall plot where multi_Hit indicates genes mutated more than once in the same sample. <bold>(H)</bold> Comparison to mutational load in a cohort of 33 cancer species already available in TCGA. <bold>(I)</bold> Distribution of antigen presentation quantities predicted by STMHCPan, STMHCPan-neo, STMHCPan+STMHCPan-neo, LRMAHPan BA, LRMAHPan AP, and LRMAHPan PS under TPM&gt;0, TPM&gt;1, and TPM&gt;2. <bold>(J)</bold> The number of Candidate neoantigen predicted by the model compared to the number of SNV mutations per sample. <bold>(K)</bold> BRAF mutation distribution and protein domain in metastatic melanoma cohort.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1478201-g004.tif"/>
</fig>
<p>Antigen presentation prediction was performed using 26 samples with available RNA expression levels employing LRMAHpan. Within the metastatic melanoma cohort, 14,462 single nucleotide variants (SNVs) were identified. Following segmentation around the mutation sites into 8-11mers, a total of 541,783 peptides were generated. The distribution of predicted peptides using STMHCPan (<xref ref-type="bibr" rid="B48">48</xref>), STMHCPan-neo, STMHCPan + STMHCPan-neo, LRMAHpan BA, LRMAHpan AP, and LRMAHpan PS was assessed under the conditions of TPM &gt; 0, TPM &gt; 1 and TPM &gt; 2 (see <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4I</bold>
</xref>). As TPM thresholds increased, the number of predicted peptides decreased. Specifically, under TPM&gt;0, LRMAHpan PS projected 8,155 presented peptides; under TPM&gt;1, 5,432 peptides were predicted; and under TPM&gt;2, 4,905 peptides were anticipated. In the prediction of presented peptides in tumor patients, combining TPM with LRMAHpan significantly reduced the false positive rate. The number of predicted novel antigens for each sample correlated with the respective SNV mutation burden (see <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4J</bold>
</xref>), suggesting that patients with a higher mutation load may derive greater benefit from immunotherapy targeting neoantigens.</p>
<p>We observed that most of the mutant peptides are unique, which may be related to the genetic diversity within the tumor and the high mutation load of melanoma. However, some shared neoantigens were detected, indicating peptide presentation across multiple samples. Specifically, LRMAHpan PS predicted 10 peptides to be presented in more than two samples (see <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>), and 116 peptides were predicted to be presented by LRMAHpan PS in more than one sample.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Shared neoantigen peptides with more than 2 samples.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Peptide</th>
<th valign="top" align="center">Sample Id</th>
<th valign="top" align="center">HGVSp_Short</th>
<th valign="top" align="center">Hugo_Symbol</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">AVYPRAGRK</td>
<td valign="top" align="center">Pt1/Pt16/Pt29</td>
<td valign="top" align="center">p.S381R</td>
<td valign="top" align="center">OAS3</td>
</tr>
<tr>
<td valign="top" align="center">SESTQENNQGY</td>
<td valign="top" align="center">Pt1/Pt4/Pt27</td>
<td valign="top" align="center">p.G444E</td>
<td valign="top" align="center">EBF2</td>
</tr>
<tr>
<td valign="top" align="center">AQVGVATY</td>
<td valign="top" align="center">Pt4/Pt6/Pt13</td>
<td valign="top" align="center">p.R381Q</td>
<td valign="top" align="center">VWA2</td>
</tr>
<tr>
<td valign="top" align="center">RAQVGVATY</td>
<td valign="top" align="center">Pt4/Pt6/Pt13</td>
<td valign="top" align="center">p.R381Q</td>
<td valign="top" align="center">VWA2</td>
</tr>
<tr>
<td valign="top" align="center">SRAQVGVATY</td>
<td valign="top" align="center">Pt4/Pt6/Pt13</td>
<td valign="top" align="center">p.R381Q</td>
<td valign="top" align="center">VWA2</td>
</tr>
<tr>
<td valign="top" align="center">KIGDFGLATEK</td>
<td valign="top" align="center">Pt6/Pt13/Pt14/Pt22/Pt32/Pt38</td>
<td valign="top" align="center">p.V600E</td>
<td valign="top" align="center">BRAF</td>
</tr>
<tr>
<td valign="top" align="center">VQDHGQPSL</td>
<td valign="top" align="center">Pt4/Pt16/Pt19</td>
<td valign="top" align="center">p.P684S/p.P653S</td>
<td valign="top" align="center">PCDHGA4/PCDHGA12</td>
</tr>
<tr>
<td valign="top" align="center">FLDPADIAA</td>
<td valign="top" align="center">Pt20/Pt28/Pt35</td>
<td valign="top" align="center">p.T315A</td>
<td valign="top" align="center">FAM160B2</td>
</tr>
<tr>
<td valign="top" align="center">FLDPADIAAL</td>
<td valign="top" align="center">Pt20/Pt28/Pt35</td>
<td valign="top" align="center">p.T315A</td>
<td valign="top" align="center">FAM160B2</td>
</tr>
<tr>
<td valign="top" align="center">ATDGGGLSEK</td>
<td valign="top" align="center">Pt5/Pt27/Pt35</td>
<td valign="top" align="center">p.E137K/p.G328E</td>
<td valign="top" align="center">PCDHB5/PCDHB6</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Notably, LRMAHpan PS identified the peptide KIGDFGLATEK, derived from the BRAF V600E mutation, in six samples. The oncogenic BRAF mutation, found in approximately 40% of melanomas, leads to sustained activation of the MAPK signaling pathway, influencing tumor cell differentiation, proliferation, and metabolism (<xref ref-type="bibr" rid="B49">49</xref>). The BRAF V600E mutation, situated within the protein tyrosine kinase domain, was detected in 7 out of 26 samples within the metastatic melanoma cohort (see <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4K</bold>
</xref>).</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>The prediction of antigen presentation is a pivotal aspect of anticipation tumor neoantigens. While many models for antigen presentation prediction predominantly concentrate on a single allele, the clinical dataset primarily consists of multi-allelic (MA) peptide sequences. As MA data continues to accumulate, direct MA antigen presentation prediction becomes feasible.</p>
<p>This study outlines strategies for applying ResNet in bioinformatics, specifically for predicting HLA class I peptide binding and presentation. By leveraging existing MA MS sequence encoding, we devised a representation conducive to integrating bioinformatics tasks with computer vision techniques. Utilizing this coding representation, we developed a ResNet-based architecture for HLA class I peptide binding prediction, which also yielded commendable results in predicting HLA class II binding. Notably, our framework enables the accurate prediction of any MA subtype. Despite being constructed with minimal data, our experimental findings on benchmark datasets demonstrate that our approach achieves state-of-the-art prediction performance across the majority of test sets compared to current models, particularly excelling on large datasets.</p>
<p>Initially, we explored data augmentation techniques to enhance the generalization ability of model. This approach was intended to increase the variability of the training data and improve performance on unseen samples. However, extensive validation revealed that the presence or absence of the data augmentation module had minimal impact on the overall performance of our predictive model. Consequently, we decided to remove the data augmentation step to streamline the computational process without sacrificing predictive accuracy.</p>
<p>Nevertheless, our model has limitations. In cases where HLA of patient type is incomplete, it requires supplementation based on known HLA typing of the patient, which may lead to some loss of accuracy. Additionally, our model&#x2019;s capacity to predict neoantigens is restricted, as our work primarily focuses on HLA class I ligand presentation without verifying ligand binding to T-cell receptors (TCR). Future research will explore the potential integration of these predictors with TCR assessments.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in onlinerepositories. The names of the repository/repositories and accession number(s) can be found in the article/<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Material</bold></xref>.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>XM: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x2013; original draft. SL: Conceptualization, Data curation, Formal analysis, Methodology, Writing &#x2013; review &amp; editing. ZY: Conceptualization, Project administration, Supervision, Writing &#x2013; review &amp; editing. ZD: Validation, Writing &#x2013; review &amp; editing. BD: Project administration, Supervision, Writing &#x2013; review &amp; editing. BS: Project administration, Supervision, Writing &#x2013; review &amp; editing. YS: Funding acquisition, Project administration, Supervision, Writing &#x2013; review &amp; editing, Methodology. ZX: Funding acquisition, Project administration, Supervision, Writing &#x2013; review &amp; editing, Formal analysis, Investigation, Resources.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by the National Natural Science Foundation of China (No. 81671807, 82072078), the Key Research &amp; Development Program of Jiangsu Province (BE2020777, SBE2020741118) and Fundamental Research Funds for the Central Universities (2242018K3DN05). Scientific Research Project of Jiangsu Health Commission(Grant No. ZDA2020012)This research work was supported by the shared service platform of data computing center of Southeast University.</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fimmu.2024.1478201/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fimmu.2024.1478201/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Williams</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>The cell biology of MHC class I antigen presentation</article-title>. <source>Tissue Antigens</source>. (<year>2002</year>) <volume>59</volume>:<fpage>3</fpage>&#x2013;<lpage>17</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1034/j.1399-0039.2002.590103.x</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Castellino</surname> <given-names>F</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>GM</given-names>
</name>
<name>
<surname>Germain</surname> <given-names>RN</given-names>
</name>
</person-group>. <article-title>Antigen presentation by MHC class II molecules: Invariant chain function, protein trafficking, and the molecular basis of diverse determinant capture</article-title>. <source>Hum Immunol</source>. (<year>1997</year>) <volume>54</volume>:<page-range>159&#x2013;69</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0198-8859(97)00078-5</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rashidi</surname> <given-names>S</given-names>
</name>
<name>
<surname>Vieira</surname> <given-names>C</given-names>
</name>
<name>
<surname>Tuteja</surname> <given-names>R</given-names>
</name>
<name>
<surname>Mansouri</surname> <given-names>R</given-names>
</name>
<name>
<surname>Ali-Hassanzadeh</surname> <given-names>M</given-names>
</name>
<name>
<surname>Muro</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Immunomodulatory potential of non-classical HLA-G in infections including COVID-19 and parasitic diseases</article-title>. <source>Biomolecules</source>. (<year>2022</year>) <volume>12</volume>(<issue>2</issue>):<fpage>257</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/biom12020257</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Zakariah</surname> <given-names>M</given-names>
</name>
<name>
<surname>Rolfo</surname> <given-names>C</given-names>
</name>
<name>
<surname>Robrecht</surname> <given-names>L</given-names>
</name>
<name>
<surname>Palaniappan</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Prediction of mycoplasma hominis proteins targeting in mitochondria and cytoplasm of host cells and their implication in prostate cancer etiology</article-title>. <source>Oncotarget</source>. (<year>2017</year>) <volume>8</volume>:<page-range>30830&#x2013;43</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.18632/oncotarget.v8i19</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khan</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Potential role of Escherichia coli DNA mismatch repair proteins in colon cancer</article-title>. <source>Crit Rev Oncol Hematol</source>. (<year>2015</year>) <volume>96</volume>:<page-range>475&#x2013;82</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.critrevonc.2015.05.002</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Imran</surname> <given-names>A</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Abul Kalam</surname> <given-names>M</given-names>
</name>
<name>
<surname>Alshamsan</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Systems biology approaches for the prediction of possible role of chlamydia pneumoniae proteins in the etiology of lung cancer</article-title>. <source>PloS One</source>. (<year>2016</year>) <volume>11</volume>:<elocation-id>e0148530</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0148530</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zakariah</surname> <given-names>M</given-names>
</name>
<name>
<surname>Malik</surname> <given-names>A</given-names>
</name>
<name>
<surname>Ola</surname> <given-names>MS</given-names>
</name>
<name>
<surname>Syed</surname> <given-names>R</given-names>
</name>
<name>
<surname>Chaudhary</surname> <given-names>AA</given-names>
</name>
<etal/>
</person-group>. <article-title>Analysis of Salmonella typhimurium Protein-Targeting in the Nucleus of Host Cells and the Implications in Colon Cancer: An in-silico Approach</article-title>. <source>Infect Drug Resist</source>. (<year>2020</year>) <volume>13</volume>:<page-range>2433&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2147/IDR.S258037</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Imran</surname> <given-names>A</given-names>
</name>
<name>
<surname>Shami</surname> <given-names>A</given-names>
</name>
<name>
<surname>Chaudhary</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Decipher the Helicobacter pylori Protein Targeting in the Nucleus of Host Cell and their Implications in Gallbladder Cancer: An insilico approach</article-title>. <source>J Cancer</source>. (<year>2021</year>) <volume>12</volume>:<page-range>7214&#x2013;22</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.7150/jca.63517</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname> <given-names>P</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>Polymorphism in clinical immunology - From HLA typing to immunogenetic profiling</article-title>. <source>J Trans Med</source>. (<year>2003</year>) <volume>1</volume>:<fpage>8</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1479-5876-1-8</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Williams</surname> <given-names>TM</given-names>
</name>
</person-group>. <article-title>Human leukocyte antigen gene polymorphism and the histocompatibility laboratory</article-title>. <source>J Mol Diagnostics</source>. (<year>2001</year>) <volume>3</volume>:<fpage>98</fpage>&#x2013;<lpage>104</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S1525-1578(10)60658-7</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rock</surname> <given-names>KL</given-names>
</name>
<name>
<surname>Reits</surname> <given-names>E</given-names>
</name>
<name>
<surname>Neefjes</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Present yourself! By MHC class I and MHC class II molecules</article-title>. <source>Trends Immunol</source>. (<year>2016</year>) <volume>37</volume>:<page-range>724&#x2013;37</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.it.2016.08.010</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yewdell</surname> <given-names>JW</given-names>
</name>
<name>
<surname>Bennink</surname> <given-names>JR</given-names>
</name>
</person-group>. <article-title>Immunodominance in major histocompatibility complex class I-restricted T lymphocyte responses</article-title>. <source>Annu Rev Immunol</source>. (<year>1999</year>) <volume>17</volume>:<fpage>51</fpage>&#x2013;<lpage>88</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev.immunol.17.1.51</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Ott</surname> <given-names>PA</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>CJ</given-names>
</name>
</person-group>. <article-title>Towards personalized, tumour-specific, therapeutic vaccines for cancer</article-title>. <source>Nat Rev Immunol</source>. (<year>2018</year>) <volume>18</volume>:<page-range>168&#x2013;82</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nri.2017.131</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peters</surname> <given-names>B</given-names>
</name>
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sette</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>T cell epitope predictions</article-title>. <source>Annu Rev Immunol</source>. (<year>2020</year>) <volume>38</volume>:<page-range>123&#x2013;45</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev-immunol-082119-124838</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhasin</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lata</surname> <given-names>S</given-names>
</name>
<name>
<surname>Raghava</surname> <given-names>GPS</given-names>
</name>
</person-group>. <article-title>TAPPred prediction of TAP-binding peptides in antigens</article-title>. <source>Methods Mol Biol (Clifton N.J.)</source>. (<year>2007</year>) <volume>409</volume>:<page-range>381&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-1-60327-118-9_28</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ke&#x15f;mir</surname> <given-names>C</given-names>
</name>
<name>
<surname>Nussbaum</surname> <given-names>AK</given-names>
</name>
<name>
<surname>Schild</surname> <given-names>H</given-names>
</name>
<name>
<surname>Detours</surname> <given-names>V</given-names>
</name>
<name>
<surname>Brunak</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Prediction of proteasome cleavage motifs by neural networks</article-title>. <source>Protein Eng</source>. (<year>2002</year>) <volume>15</volume>:<page-range>287&#x2013;96</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/protein/15.4.287</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lundegaard</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lund</surname> <given-names>O</given-names>
</name>
<name>
<surname>Kesmir</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>The role of the proteasome in generating cytotoxic T-cell epitopes: insights obtained from improved predictions of proteasomal cleavage</article-title>. <source>Immunogenetics</source>. (<year>2005</year>) <volume>57</volume>:<fpage>33</fpage>&#x2013;<lpage>41</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00251-005-0781-7</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peters</surname> <given-names>B</given-names>
</name>
<name>
<surname>Bulik</surname> <given-names>S</given-names>
</name>
<name>
<surname>Tampe</surname> <given-names>R</given-names>
</name>
<name>
<surname>Van Endert</surname> <given-names>PM</given-names>
</name>
<name>
<surname>Holzh&#xfc;tter</surname> <given-names>HG</given-names>
</name>
</person-group>. <article-title>Identifying MHC class I epitopes by predicting the TAP transport efficiency of epitope precursors</article-title>. <source>J Immunol</source>. (<year>2003</year>) <volume>171</volume>:<page-range>1741&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.171.4.1741</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Larsen</surname> <given-names>MV</given-names>
</name>
<name>
<surname>Lundegaard</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lamberth</surname> <given-names>K</given-names>
</name>
<name>
<surname>Buus</surname> <given-names>S</given-names>
</name>
<name>
<surname>Brunak</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lund</surname> <given-names>O</given-names>
</name>
<etal/>
</person-group>. <article-title>An integrative approach to CTL epitope prediction: a combined algorithm integrating MHC class I binding, TAP transport efficiency, and proteasomal cleavage predictions</article-title>. <source>Eur J Immunol</source>. (<year>2005</year>) <volume>35</volume>:<page-range>2295&#x2013;303</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/(ISSN)1521-4141</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stranzl</surname> <given-names>T</given-names>
</name>
<name>
<surname>Larsen</surname> <given-names>MV</given-names>
</name>
<name>
<surname>Lundegaard</surname> <given-names>C</given-names>
</name>
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>NetCTLpan: pan-specific MHC class I pathway epitope predictions</article-title>. <source>Immunogenetics</source>. (<year>2010</year>) <volume>62</volume>:<page-range>357&#x2013;68</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00251-010-0441-4</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tenzer</surname> <given-names>S</given-names>
</name>
<name>
<surname>Peters</surname> <given-names>B</given-names>
</name>
<name>
<surname>Bulik</surname> <given-names>S</given-names>
</name>
<name>
<surname>Schoor</surname> <given-names>O</given-names>
</name>
<name>
<surname>Lemmel</surname> <given-names>C</given-names>
</name>
<name>
<surname>Schatz</surname> <given-names>MM</given-names>
</name>
<etal/>
</person-group>. <article-title>Modeling the MHC class I pathway by combining predictions of proteasomal cleavage, TAP transport and MHC class I binding</article-title>. <source>Cell Mol Life Sci</source>. (<year>2005</year>) <volume>62</volume>:<page-range>1025&#x2013;37</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00018-005-4528-2</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>O'Donnell</surname> <given-names>TJ</given-names>
</name>
<name>
<surname>Rubinsteyn</surname> <given-names>A</given-names>
</name>
<name>
<surname>Laserson</surname> <given-names>U</given-names>
</name>
</person-group>. <article-title>MHCflurry 2.0: improved pan-allele prediction of MHC class I-presented peptides by incorporating antigen processing</article-title>. <source>Cell Syst</source>. (<year>2020</year>) <volume>11</volume>:<fpage>42</fpage>&#x2013;<lpage>48.e47</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cels.2020.09.001</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bulik-Sullivan</surname> <given-names>B</given-names>
</name>
<name>
<surname>Busby</surname> <given-names>J</given-names>
</name>
<name>
<surname>Palmer</surname> <given-names>CD</given-names>
</name>
<name>
<surname>Davis</surname> <given-names>MJ</given-names>
</name>
<name>
<surname>Murphy</surname> <given-names>T</given-names>
</name>
<name>
<surname>Clark</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Deep learning using tumor HLA peptide mass spectrometry datasets improves neoantigen identification</article-title>. <source>Nat Biotechnol</source>. (<year>2019</year>) <volume>37</volume>:<fpage>55</fpage>&#x2013;<lpage>63</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nbt.4313</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>C</given-names>
</name>
<name>
<surname>van den Hengel</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Wider or deeper: revisiting the resNet model for visual recognition</article-title>. <source>Pattern Recognition</source>. (<year>2019</year>) <volume>90</volume>:<page-range>119&#x2013;33</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.patcog.2019.01.006</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Jiao</surname> <given-names>L</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>F</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Residual spectral-spatial attention network for hyperspectral image classification</article-title>. <source>IEEE Trans Geosci Remote Sens PP</source>. (<year>2020</year>) <volume>59</volume>:<fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TGRS.2020.2994057</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Kong</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>B</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Residual dense network for image super-resolution</article-title>. <source>Proc IEEE Conf Comput Vision Pattern recognition</source>. (<year>2018</year>) <volume>43</volume>:<page-range>2472&#x2013;81</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2018.00262</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hou</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>D</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Coordinate attention for efficient mobile network design</article-title>. <source>Proc IEEE/CVF Conf Comput Vision Pattern recognition</source>. (<year>2021</year>), <page-range>13713&#x2013;22</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01350</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>M-H</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>T-X</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J-J</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z-N</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>P-T</given-names>
</name>
<name>
<surname>Mu</surname> <given-names>T-J</given-names>
</name>
<etal/>
</person-group>. <article-title>Attention mechanisms in computer vision: A survey</article-title>. <source>Comput Visual media</source>. (<year>2022</year>) <volume>8</volume>:<page-range>331&#x2013;68</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s41095-022-0271-y</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>L</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S</given-names>
</name>
<name>
<surname>Bai</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Miao</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Review of image classification algorithms based on convolutional neural networks</article-title>. <source>Remote Sens</source>. (<year>2021</year>) <volume>13</volume>:<fpage>4712</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs13224712</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>B</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>J</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Diversified visual attention networks for fine-grained object classification</article-title>. <source>IEEE Trans Multimedia</source>. (<year>2017</year>) <volume>19</volume>:<page-range>1245&#x2013;56</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TMM.2017.2648498</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reynisson</surname> <given-names>B</given-names>
</name>
<name>
<surname>Alvarez</surname> <given-names>B</given-names>
</name>
<name>
<surname>Paul</surname> <given-names>S</given-names>
</name>
<name>
<surname>Peters</surname> <given-names>B</given-names>
</name>
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>NetMHCpan-4.1 and NetMHCIIpan-4.0: improved predictions of MHC antigen presentation by concurrent motif deconvolution and integration of MS MHC eluted ligand data</article-title>. <source>Nucleic Acids Res</source>. (<year>2020</year>) <volume>48</volume>:<page-range>W449&#x2013;54</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkaa379</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>A transformer-based model to predict peptide&#x2013;HLA class I binding and optimize mutated peptides for vaccine design</article-title>. <source>Nat Mach Intell</source>. (<year>2022</year>) <volume>4</volume>:<page-range>300&#x2013;11</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s42256-022-00459-7</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gfeller</surname> <given-names>D</given-names>
</name>
<name>
<surname>Guillaume</surname> <given-names>P</given-names>
</name>
<name>
<surname>Michaux</surname> <given-names>J</given-names>
</name>
<name>
<surname>Pak</surname> <given-names>H-S</given-names>
</name>
<name>
<surname>Daniel</surname> <given-names>RT</given-names>
</name>
<name>
<surname>Racle</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>The length distribution and multiple specificity of naturally presented HLA-I ligands</article-title>. <source>J Immunol</source>. (<year>2018</year>) <volume>201</volume>:<page-range>3705&#x2013;16</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.1800914</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cerami</surname> <given-names>E</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>J</given-names>
</name>
<name>
<surname>Dogrusoz</surname> <given-names>U</given-names>
</name>
<name>
<surname>Gross</surname> <given-names>BE</given-names>
</name>
<name>
<surname>Sumer</surname> <given-names>SO</given-names>
</name>
<name>
<surname>Aksoy</surname> <given-names>BA</given-names>
</name>
<etal/>
</person-group>. <article-title>The cBio cancer genomics portal: an open platform for exploring multidimensional cancer genomics data</article-title>. <source>Cancer Discovery</source>. (<year>2012</year>) <volume>2</volume>:<page-range>401&#x2013;4</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/2159-8290.CD-12-0095</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname> <given-names>J</given-names>
</name>
<name>
<surname>Aksoy</surname> <given-names>BA</given-names>
</name>
<name>
<surname>Dogrusoz</surname> <given-names>U</given-names>
</name>
<name>
<surname>Dresdner</surname> <given-names>G</given-names>
</name>
<name>
<surname>Gross</surname> <given-names>B</given-names>
</name>
<name>
<surname>Sumer</surname> <given-names>SO</given-names>
</name>
<etal/>
</person-group>. <article-title>Integrative analysis of complex cancer genomics and clinical profiles using the cBioPortal</article-title>. <source>Sci Signaling</source>. (<year>2013</year>) <volume>6</volume>:<page-range>pl1&#x2013;1</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/scisignal.2004088</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hoof</surname> <given-names>I</given-names>
</name>
<name>
<surname>Peters</surname> <given-names>B</given-names>
</name>
<name>
<surname>Sidney</surname> <given-names>J</given-names>
</name>
<name>
<surname>Pedersen</surname> <given-names>LE</given-names>
</name>
<name>
<surname>Sette</surname> <given-names>A</given-names>
</name>
<name>
<surname>Lund</surname> <given-names>O</given-names>
</name>
<etal/>
</person-group>. <article-title>NetMHCpan, a method for MHC class I binding prediction beyond humans</article-title>. <source>Immunogenetics</source>. (<year>2009</year>) <volume>61</volume>:<fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00251-008-0341-z</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karosiene</surname> <given-names>E</given-names>
</name>
<name>
<surname>Rasmussen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Blicher</surname> <given-names>T</given-names>
</name>
<name>
<surname>Lund</surname> <given-names>O</given-names>
</name>
<name>
<surname>Buus</surname> <given-names>S</given-names>
</name>
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>NetMHCIIpan-3.0, a common pan-specific MHC class II prediction method including all three human MHC class II isotypes, HLA-DR, HLA-DP and HLA-DQ</article-title>. <source>Immunogenetics</source>. (<year>2013</year>) <volume>65</volume>:<page-range>711&#x2013;24</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00251-013-0720-y</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Foret</surname> <given-names>P</given-names>
</name>
<name>
<surname>Kleiner</surname> <given-names>A</given-names>
</name>
<name>
<surname>Mobahi</surname> <given-names>H</given-names>
</name>
<name>
<surname>Neyshabur</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Sharpness-aware minimization for efficiently improving generalization</article-title>. <source>ArXiv</source>. (<year>2020</year>) abs/2010.01412. Available online at: <uri xlink:href="https://api.semanticscholar.org/CorpusID:222134093">https://api.semanticscholar.org/CorpusID:222134093</uri>.</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bottou</surname> <given-names>L</given-names>
</name>
<name>
<surname>Curtis</surname> <given-names>FE</given-names>
</name>
<name>
<surname>Nocedal</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Optimization methods for large-scale machine learning</article-title>. <source>SIAM Rev</source>. (<year>2018</year>) <volume>60</volume>:<fpage>223</fpage>&#x2013;<lpage>311</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1137/16M1080173</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>W</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>Bidirectional LSTM-CRF models for sequence tagging</article-title>. <source>arXiv preprint arXiv:1508.01991</source>. (<year>2015</year>).</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Smirnov</surname> <given-names>EA</given-names>
</name>
<name>
<surname>Timoshenko</surname> <given-names>DM</given-names>
</name>
<name>
<surname>Andrianov</surname> <given-names>SN</given-names>
</name>
</person-group>. <article-title>Comparison of regularization methods for ImageNet classification with deep convolutional neural networks</article-title>. <source>AASRI Procedia</source> (<year>2013</year>) <volume>6</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.aasri.2014.05.013</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Revisiting spatial dropout for regularizing convolutional neural networks</article-title>. <source>Multimedia Tools Appl</source>. (<year>2020</year>) <volume>79</volume>:<page-range>34195&#x2013;207</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11042-020-09054-7</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>XY</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>ZH</given-names>
</name>
</person-group>. <article-title>Exploratory undersampling for class-imbalance learning</article-title>. <source>IEEE Trans Syst Man Cybernetics Part B</source>. (<year>2009</year>) <volume>39</volume>:<page-range>539&#x2013;50</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TSMCB.2008.2007853</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shao</surname> <given-names>XM</given-names>
</name>
<name>
<surname>Bhattacharya</surname> <given-names>R</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Sivakumar</surname> <given-names>IKA</given-names>
</name>
<name>
<surname>Tokheim</surname> <given-names>C</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>High-throughput prediction of MHC class I and II neoantigens with MHCnuggets</article-title>. <source>Cancer Immunol Res</source>. (<year>2020</year>) <volume>8</volume>:<fpage>396</fpage>&#x2013;<lpage>408</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/2326-6066.CIR-19-0464</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dao</surname> <given-names>T</given-names>
</name>
<name>
<surname>Klatt</surname> <given-names>MG</given-names>
</name>
<name>
<surname>Korontsvit</surname> <given-names>T</given-names>
</name>
<name>
<surname>Mun</surname> <given-names>SS</given-names>
</name>
<name>
<surname>Guzman</surname> <given-names>S</given-names>
</name>
<name>
<surname>Mattar</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Impact of tumor heterogeneity and microenvironment in identifying neoantigens in a patient with ovarian cancer</article-title>. <source>Cancer Immunology Immunotherapy</source>. (<year>2021</year>) <volume>70</volume>:<page-range>1189&#x2013;202</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00262-020-02764-9</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barahona</surname> <given-names>R</given-names>
</name>
<name>
<surname>Maria</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Deep learning for sentiment analysis</article-title>. <source>Lang Linguistics Compass</source>. (<year>2016</year>) <volume>10</volume>:<page-range>205&#x2013;12</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/lnc3.12228</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mayakonda</surname> <given-names>A</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>DC</given-names>
</name>
<name>
<surname>Assenov</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Plass</surname> <given-names>C</given-names>
</name>
<name>
<surname>Koeffler</surname> <given-names>HP</given-names>
</name>
</person-group>. <article-title>Maftools: efficient and comprehensive analysis of somatic variants in cancer</article-title>. <source>Genome Res</source>. (<year>2018</year>) <volume>28</volume>:<page-range>1747&#x2013;56</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/gr.239244.118</pub-id>
</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S</given-names>
</name>
<name>
<surname>Mi</surname> <given-names>X</given-names>
</name>
<name>
<surname>Shao</surname> <given-names>B</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>STMHCpan, an accurate Star-Transformer-based extensible framework for predicting MHC I allele binding peptides</article-title>. <source>Brief Bioinform</source>. (<year>2023</year>) <volume>24</volume>(<issue>3</issue>):<fpage>bbad164</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bib/bbad164</pub-id>
</citation>
</ref>
<ref id="B49">
<label>49</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Long</surname> <given-names>GV</given-names>
</name>
<name>
<surname>Menzies</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Nagrial</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Haydu</surname> <given-names>LE</given-names>
</name>
<name>
<surname>Hamilton</surname> <given-names>AL</given-names>
</name>
<name>
<surname>Mann</surname> <given-names>GJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Prognostic and clinicopathologic associations of oncogenic BRAF in metastatic melanoma</article-title>. <source>J Clin Oncol</source>. (<year>2011</year>) <volume>29</volume>:<page-range>1239&#x2013;46</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/JCO.2010.32.4327</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>