<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2024.1394593</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>MONET: a database for prediction of neoantigens derived from microsatellite loci</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Deng</surname>
<given-names>Nan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2212588"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sinha</surname>
<given-names>Krishna M.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1169779"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Vilar</surname>
<given-names>Eduardo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/676939"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Clinical Cancer Prevention, The University of Texas MD Anderson Cancer Center</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Gastrointestinal Medical Oncology, The University of Texas MD Anderson Cancer Center</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Clinical Cancer Genetics Program, The University of Texas MD Anderson Cancer Center</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Zlatko Trajanoski, Medical University of Innsbruck, Austria</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Martin L&#xf6;wer, Translationale Onkologie an der Universit&#xe4;tsmedizin der Johannes Gutenberg-Universit&#xe4;t Mainz, Germany</p>
<p>Cansu Cimen Bozkus, Icahn School of Medicine at Mount Sinai, United States</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Nan Deng, <email xlink:href="mailto:Ndeng1@mdanderson.org">Ndeng1@mdanderson.org</email>; Eduardo Vilar, <email xlink:href="mailto:EVilar@mdanderson.org">EVilar@mdanderson.org</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1394593</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>03</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Deng, Sinha and Vilar</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Deng, Sinha and Vilar</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Microsatellite instability (MSI) secondary to mismatch repair (MMR) deficiency is characterized by insertions and deletions (indels) in short DNA sequences across the genome. These indels can generate neoantigens, which are ideal targets for precision immune interception. However, current neoantigen databases lack information on neoantigens arising from coding microsatellites. To address this gap, we introduce The MicrOsatellite Neoantigen Discovery Tool (MONET).</p>
</sec>
<sec>
<title>Method</title>
<p>MONET identifies potential mutated tumor-specific neoantigens (neoAgs) by predicting frameshift mutations in coding microsatellite sequences of the human genome. Then MONET annotates these neoAgs with key features such as binding affinity, stability, expression, frequency, and potential pathogenicity using established algorithms, tools, and public databases. A user-friendly web interface (<uri xlink:href="https://monet.mdanderson.org/">https://monet.mdanderson.org/</uri>) facilitates access to these predictions.</p>
</sec>
<sec>
<title>Results</title>
<p>MONET predicts over 4 million and 15 million Class I and Class II potential frameshift neoAgs, respectively. Compared to existing databases, MONET demonstrates superior coverage (&gt;85% vs. &lt;25%) using a set of experimentally validated neoAgs.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>MONET is a freely available, user-friendly web tool that leverages publicly available resources to identify neoAgs derived from microsatellite loci. This systems biology approach empowers researchers in the field of precision immune interception.</p>
</sec>
</abstract>
<kwd-group>
<kwd>neoantigen</kwd>
<kwd>microsatellite</kwd>
<kwd>Lynch syndrome</kwd>
<kwd>mismatch repair</kwd>
<kwd>somatic mutation</kwd>
<kwd>indels</kwd>
</kwd-group>
<contract-num rid="cn001">R01CA260761, R01CA257375 , U01 CA233056 , P50 CA221707 , P30 CA016672</contract-num>
<contract-sponsor id="cn001">National Institutes of Health<named-content content-type="fundref-id">10.13039/100000002</named-content>
</contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="30"/>
<page-count count="9"/>
<word-count count="3485"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Cancer Immunity and Immunotherapy</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>Microsatellite instability (MSI) is caused by the accumulation of insertions and deletions (indels) in short-segment DNA sequences of mono-, di-, tri-nucleotide, and longer repeats known as microsatellites due to mismatch repair (MMR) deficiency. MMR deficiency is secondary to inactivating mutations in one of the four MMR genes (<italic>MLH1, MSH2, MHS6</italic>, and <italic>PMS2</italic>) or epigenetic silencing of <italic>MLH1</italic> (sporadic MMR deficiency) (<xref ref-type="bibr" rid="B1">1</xref>). These indels lead to frameshift mutations, thus resulting in the generation of mutated neoantigens (neoAgs) that are unique to tumor cells and highly unlikely to be found in normal cells. Unlike non-synonymous single-nucleotide variants (SNVs), which are mutations that typically generate neoAgs with only one altered amino acid(s), frameshift mutations typically generate completely different amino acid sequences. These frameshifted sequences have low probabilities of being tolerated by the host&#x2019;s immune system. These mutated frameshift proteins possess intrinsic immunogenicity and are, therefore, attractive targets for cancer interception and therapy (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B3">3</xref>).</p>
<p>Currently, there are several epitope databases available to facilitate <italic>in silico</italic> vaccine design (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). The Immune Epitope Database (IEDB) (<xref ref-type="bibr" rid="B4">4</xref>) is a globally accessible gateway to experimentally validated immune epitopes, while other databases, such as AntiJen (<xref ref-type="bibr" rid="B5">5</xref>) and caped (<xref ref-type="bibr" rid="B6">6</xref>) focus on curated cancer epitopes from research manuscripts. In addition, the GNIFdb (<xref ref-type="bibr" rid="B7">7</xref>) and TSNAdb databases (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B9">9</xref>) utilize computational approaches to predict putative neoAg based on high-frequency mutations detected in cancers. However, there is currently no antigen database dedicated to neoAgs derived from microsatellite tracts. Tumors displaying high levels of MSI (MSI-H) may harbor indels in up to 80% of microsatellite loci (<xref ref-type="bibr" rid="B10">10</xref>), thus suggesting that coding MSI could generate a significant number of potential neoAg candidates with high degree of immunogenicity. Leveraging this knowledge gap, we developed a new database named MicrOsatellite NEoantigen Discovery Tool (MONET) that focuses on the prediction of putative neoAgs derived from MSI in cancers.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Available antigen databases.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Database</th>
<th valign="top" align="left">URL</th>
<th valign="top" align="left">Type</th>
<th valign="top" align="left">Target</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">GNIFdb</td>
<td valign="top" align="left">
<ext-link ext-link-type="uri" xlink:href="http://www.oncoimmunobank.cn/index.php">http://www.oncoimmunobank.cn/index.php</ext-link>
</td>
<td valign="top" align="left">Computational</td>
<td valign="top" align="left">Glioma</td>
</tr>
<tr>
<td valign="top" align="left">IEDB</td>
<td valign="top" align="left">
<ext-link ext-link-type="uri" xlink:href="https://www.iedb.org/">https://www.iedb.org/</ext-link>
</td>
<td valign="top" align="left">Curated</td>
<td valign="top" align="left">General</td>
</tr>
<tr>
<td valign="top" align="left">TSNAdb v2.0</td>
<td valign="top" align="left">
<ext-link ext-link-type="uri" xlink:href="https://pgx.zju.edu.cn/tsnadb/">https://pgx.zju.edu.cn/tsnadb/</ext-link>
</td>
<td valign="top" align="left">Computational</td>
<td valign="top" align="left">Pan cancer</td>
</tr>
<tr>
<td valign="top" align="left">dbPepNeo2.0</td>
<td valign="top" align="left">
<ext-link ext-link-type="uri" xlink:href="http://119.3.70.71/dbPepNeo2/home.html">http://119.3.70.71/dbPepNeo2/home.html</ext-link>
</td>
<td valign="top" align="left">Curated</td>
<td valign="top" align="left">Pan cancer</td>
</tr>
<tr>
<td valign="top" align="left">CAD v1.0</td>
<td valign="top" align="left">
<ext-link ext-link-type="uri" xlink:href="http://cad.bio-it.cn/">http://cad.bio-it.cn/</ext-link>
</td>
<td valign="top" align="left">Curated</td>
<td valign="top" align="left">Pan Cancer</td>
</tr>
<tr>
<td valign="top" align="left">NeoPeptide</td>
<td valign="top" align="left">Not availible</td>
<td valign="top" align="left">Curated</td>
<td valign="top" align="left">Pan Cancer</td>
</tr>
<tr>
<td valign="top" align="left">NEPdb</td>
<td valign="top" align="left">
<ext-link ext-link-type="uri" xlink:href="http://nep.whu.edu.cn/">http://nep.whu.edu.cn/</ext-link>.</td>
<td valign="top" align="left">Curated</td>
<td valign="top" align="left">Pan Cancer</td>
</tr>
<tr>
<td valign="top" align="left">TANTIGEN 2.0</td>
<td valign="top" align="left">
<ext-link ext-link-type="uri" xlink:href="http://projects.met-hilab.org/tadb">http://projects.met-hilab.org/tadb</ext-link>
</td>
<td valign="top" align="left">Curated</td>
<td valign="top" align="left">Pan Cancer</td>
</tr>
<tr>
<td valign="top" align="left">AntiJen</td>
<td valign="top" align="left">
<ext-link ext-link-type="uri" xlink:href="http://www.ddg-pharmfac.net/antijen/AntiJen/antijenhomepage.htm">http://www.ddg-pharmfac.net/antijen/AntiJen/antijenhomepage.htm</ext-link>
</td>
<td valign="top" align="left">Curated</td>
<td valign="top" align="left">General</td>
</tr>
<tr>
<td valign="top" align="left">caped</td>
<td valign="top" align="left">
<ext-link ext-link-type="uri" xlink:href="https://caped.icp.ucl.ac.be/">https://caped.icp.ucl.ac.be/</ext-link>
</td>
<td valign="top" align="left">Curated</td>
<td valign="top" align="left">General</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Here, we introduce the MONET database, which includes all possible neoAg derived from microsatellite loci present in the human reference genome. To generate MONET, we used computational algorithms to predict neoAg-derived epitopes with high affinity for CD8<sup>+</sup> and CD4<sup>+</sup> T cells that are specific to a variety of MHC-I and MHC-II alleles, respectively. We also evaluated the binding stability and foreignness of predicted epitopes, which are crucial factors in assessing their immunogenicity. Furthermore, we integrated data on gene expression levels of neoAgs in different tumor types leveraging the TCGA database, and mutation allele frequencies from public databases in order to provide a comprehensive immunogenicity and population coverage. Our work demonstrates that MONET has excellent coverage and outperforms other available antigen databases when tested against a set of verified epitopes derived from microsatellites. In addition, a user-friendly web-interface has been implemented and housed at <ext-link ext-link-type="uri" xlink:href="https://monet.mdanderson.org/">https://monet.mdanderson.org/</ext-link>, where users can query candidate target genes and obtain curated lists of potential neoAg epitopes based on their corresponding MHC alleles without the need for complex computational efforts.</p>
</sec>
<sec id="s2">
<title>Methods</title>
<sec id="s3_1">
<title>Generation of potential frameshift neoantigens</title>
<p>To generate potential frameshift neoAg derived from microsatellite loci, we utilized MSIsensor2 (<ext-link ext-link-type="uri" xlink:href="https://github.com/niu-lab/msisensor2">https://github.com/niu-lab/msisensor2</ext-link>) (<xref ref-type="bibr" rid="B11">11</xref>) to scan the human reference genome (GRCh38) for the identification of all microsatellite loci. Short nucleotide repeats exceeding five units were identified as microsatellites. For larger repeated motifs (ranging from 2 to 5 base pairs), a minimum repeat number of three was used. Only microsatellites within protein-coding regions were retained to generate potential mutant proteins. Insertions/deletions differing by a multiple of 3 will share the same reading frame, resulting in identical downstream sequences, so that we can generate two types of downstream frameshift sequences for each identified microsatellite: 3n+1 and 3n+2 shifts of nucleotide bases, where n is an integer (-3, -2, -1, 0, 1, 2, 3, and so on). For example, in the 3n+1 series, two nucleotide deletions (n = -1 and 3n+1 = -2) in the sequence will share the same reading frame with one nucleotide insertion (n = 0, 3n+1 = 1), and four nucleotide insertions (n= 1 3n+1 = 4) and so on. This applies similarly to the 3n+2 series. Therefore, to efficiently represent these frameshift mutations, we introduce two <italic>in silico</italic> variants for each microsatellite: a two-nucleotide deletion (n = -1, 3n+1 = -2) and a one-nucleotide deletion (n = -1, 3n+2 = -1). These variants encompass the spectrum of frameshift mutations within each series and generate entirely distinct downstream amino acid sequences compared to the wild-type sequence, thus making them ideal neoAg candidates.</p>
</sec>
<sec id="s3_2">
<title>Putative neoantigen epitopes</title>
<p>It is important to note the significant complexity of potential mutant amino acid sequences generated at the junction region, where the wild-type and mutant sequences meet within the microsatellite region. The size and location of indels at this junction significantly impact the resulting mutant sequence. However, based on our previous experimental data (<xref ref-type="bibr" rid="B10">10</xref>), very few (&lt;5%) verified neoAg epitopes originate from these junction regions. Therefore, we excluded the sequences around the junction at which the frameshifted amino acid occurred to prevent this complexity. Then, putative neoAg epitopes were predicted towards a panel of higher frequency MHC-Class I (<xref ref-type="bibr" rid="B12">12</xref>) and MHC-Class II (<xref ref-type="bibr" rid="B13">13</xref>) alleles (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>) providing coverage for 97% and 99% of the general population, respectively. Various algorithms such as MHCflurry (<xref ref-type="bibr" rid="B14">14</xref>), MHCnuggets (<xref ref-type="bibr" rid="B15">15</xref>), NetMHC (<xref ref-type="bibr" rid="B16">16</xref>), PickPocket (<xref ref-type="bibr" rid="B17">17</xref>), SMM-align (<xref ref-type="bibr" rid="B18">18</xref>), NNalign (<xref ref-type="bibr" rid="B19">19</xref>) that are implemented in pVACtools (<xref ref-type="bibr" rid="B20">20</xref>) were employed to predict neoAgs. Any epitope and allele pairs with a binding affinity of IC<sub>50</sub> &lt;50 nM in any algorithm were considered potential epitopes for subsequent processes. We predicted the binding stability for Class I epitopes using NetMHCstab (<xref ref-type="bibr" rid="B21">21</xref>) and assessed foreignness to the human proteome using antigen.garnish (<ext-link ext-link-type="uri" xlink:href="https://github.com/andrewrech/antigen.garnish">https://github.com/andrewrech/antigen.garnish</ext-link>) (<xref ref-type="bibr" rid="B22">22</xref>). Also, other characteristics that could play an important role in epitope selection such as terminal amino acids and the Gravy score (average hydropathy) were annotated.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>MHC Alleles used to predict putative neoantigens.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Class I Allele tested (n=27)</th>
<th valign="top" align="left">Class II Allele tested (n=27)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">HLA-A*01:01<break/>HLA-A*02:01<break/>HLA-A*02:03<break/>HLA-A*02:06<break/>HLA-A*03:01<break/>HLA-A*11:01<break/>HLA-A*23:01<break/>HLA-A*24:02<break/>HLA-A*26:01<break/>HLA-A*30:01<break/>HLA-A*30:02<break/>HLA-A*31:01<break/>HLA-A*32:01<break/>HLA-A*33:01<break/>HLA-A*68:01<break/>HLA-A*68:02<break/>HLA-B*07:02<break/>HLA-B*08:01<break/>HLA-B*15:01<break/>HLA-B*35:01<break/>HLA-B*40:01<break/>HLA-B*44:02<break/>HLA-B*44:03<break/>HLA-B*51:01<break/>HLA-B*53:01<break/>HLA-B*57:01<break/>HLA-B*58:01</td>
<td valign="top" align="left">DRB1*01:01<break/>DRB1*03:01<break/>DRB1*04:01<break/>DRB1*04:05<break/>DRB1*07:01<break/>DRB1*08:02<break/>DRB1*09:01<break/>DRB1*11:01<break/>DRB1*12:01<break/>DRB1*13:02<break/>DRB1*15:01<break/>DRB3*01:01<break/>DRB3*02:02<break/>DRB4*01:01<break/>DRB5*01:01<break/>DQA1*05:01-DQB1*02:01<break/>DQA1*05:01-DQB1*03:01<break/>DQA1*03:01-DQB1*03:02<break/>DQA1*04:01-DQB1*04:02<break/>DQA1*01:01-DQB1*05:01<break/>DQA1*01:02-DQB1*06:02<break/>DPA1*02:01-DPB1*01:01<break/>DPA1*01:03-DPB1*02:01<break/>DPA1*01:03-DPB1*04:01<break/>DPA1*03:01-DPB1*04:02<break/>DPA1*02:01-DPB1*05:01<break/>DPA1*02:01-DPB1*14:01</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_3">
<title>Annotation of neoantigens</title>
<p>To gain a deeper understanding of the potential epitopes, we also annotated the corresponding mutations using the Ensembl Variant Effect Predictor (VEP) (<xref ref-type="bibr" rid="B23">23</xref>). Annotations include associated gene names, IDs, genomic coordinates, and among others. Another critical factor for the effectiveness of neoAgs being presented by MHC is their expression level (<xref ref-type="bibr" rid="B24">24</xref>&#x2013;<xref ref-type="bibr" rid="B26">26</xref>). Higher expression levels will increase the probability of these epitopes being presented by MHC molecules. We integrated the expression levels of the corresponding genes generating the neoAgs from the Cancer Genome Atlas Program (TCGA, <ext-link ext-link-type="uri" xlink:href="https://www.cancer.gov/tcga">https://www.cancer.gov/tcga</ext-link>) project by using all datasets that contain pairs of normal and tumor. Additionally, for the dataset with MSI status information, we distinguished and listed the MSI-H and MSS groups separately. Differentially expressed genes (Benjamini-Hochberg adjusted p-value &lt; 0.05) between normal and cancer tissues in each of the different cancer data set were labelled. Furthermore, we annotated the frequency of the mutation and associated phenotypes using the dbSNP (RRID:SCR_002338) (<xref ref-type="bibr" rid="B27">27</xref>) and ClinVar (RRID:SCR_006169) database (<xref ref-type="bibr" rid="B28">28</xref>). This annotation provided valuable insights into the frequency and clinical implications of the mutations from where the neoAgs are derived. Moreover, we included evidence of experimentally confirmed epitopes using IEDB (RRID:SCR_006604), which contributes to the reliability and validity of the identified epitopes.</p>
</sec>
<sec id="s3_4">
<title>MONET website infrastructure</title>
<p>We constructed the website using a microservices architecture with Docker. The backend database (MySQL) was employed to store and manage data. Express.js served as the middleware, responsible for translating HTTP requests into MySQL queries. Vue.js was utilized to develop the user interface. Nginx served as the HTTP server communicating between the containers hosting the frontend and backend components of the website.</p>
</sec>
<sec id="s3_5">
<title>Comparisons of the epitope coverage derived from microsatellite loci across various databases</title>
<p>We selected 100 epitopes derived from mutations in microsatellite loci among the top-ranked based on their predicted immunogenicity (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>) in LS patients and showed that 65 out of 100 predicted neoAg candidates were validated for their immunogenicity using <italic>in vitro</italic> ELISPOT assays&#xa0;(<xref ref-type="bibr" rid="B10">10</xref>). These peptides were used to assess the coverage across&#xa0;different databases. Specifically, on April 9<sup>th</sup>, 2024, we individually searched the 65 validated immunogenic epitopes within MONET and the 10 databases listed in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. Then, we evaluated the number of epitopes recorded in each database. The services of dbPepNeo, NeoPeptide, and TANTIGEN databases were unavailable at the time of our research. Therefore, we reported results from MONET and the other 7 databases.</p>
</sec>
<sec id="s3_6">
<title>Data Availability</title>
<p>Public data analyzed in this study was obtained from multiple sources including: <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/genome/">https://www.ncbi.nlm.nih.gov/genome/</ext-link>, <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/snp/">https://www.ncbi.nlm.nih.gov/snp/</ext-link>, <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/clinvar/">https://www.ncbi.nlm.nih.gov/clinvar/</ext-link>, <ext-link ext-link-type="uri" xlink:href="https://www.iedb.org/">https://www.iedb.org/</ext-link>, <ext-link ext-link-type="uri" xlink:href="https://www.cancer.gov/tcga">https://www.cancer.gov/tcga</ext-link>. The epitope prediction data in this study are available at <ext-link ext-link-type="uri" xlink:href="https://monet.mdanderson.org">https://monet.mdanderson.org</ext-link>.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s4_1">
<title>Prediction of frameshift neoantigens in microsatellites</title>
<p>The overall data processing is depicted in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. We identified a total of 34,067,744 microsatellite loci in the human genome GRCh38. Among them, only 3,934,634 microsatellites were located in protein-coding regions (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>). After eliminating duplicated sequences, we obtained 492,578 unique frameshift mutation neoAg sequences (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>). From these sequences, we identified 4,449,128 MHC Class I and 15,589,846 MHC Class II potential epitopes. The number of epitopes binding to MHC Class I alleles for each allele ranged from approximately 6,000 to 800,000 (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2A</bold>
</xref>), while binders to Class II alleles displayed a higher number of predicted epitopes ranging from approximately 5,000 to 6 million (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2B</bold>
</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Schema of MONET. Microsatellite regions on the human reference genome GRCh38 were determined by <italic>MSIsensor2</italic>. The potential neoAg epitopes against high-frequency human MHC molecules were determined by <italic>pVacbind</italic>. The selected potential neoAgs were then annotated with other information, such as binding stability, experimental evidence, and frequency in populations, which will be useful information for vaccine design. Finally, we constructed a user-friendly interface by <italic>Vue.js</italic> to help users access our epitope database.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394593-g001.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Statistics of the MONET database.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">
</th>
<th valign="top" align="left">Count</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">
<bold>Microsatellites on reference genome</bold>
</td>
<td valign="top" align="left">34,067,744</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Microsatellites on exon regions</bold>
</td>
<td valign="top" align="left">3,934,634</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Genes with microsatellites on exon regions</bold>
</td>
<td valign="top" align="left">18,591</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Potential mutations</bold>
</td>
<td valign="top" align="left">11,803,902</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>No duplicated downstream sequence</bold>
</td>
<td valign="top" align="left">492,578</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Possible Class I epitopes (best affinity &lt;50nM)</bold>
</td>
<td valign="top" align="left">4,449,128</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Possible Class II epitopes (best affinity &lt;50nM)</bold>
</td>
<td valign="top" align="left">15,589,846</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Mutations in dbSNP</bold>
</td>
<td valign="top" align="left">542,442</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Mutations in ClinVar</bold>
</td>
<td valign="top" align="left">13,289</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Iedb experiment verified epitope</bold>
</td>
<td valign="top" align="left">65,220</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Iedb experiment verified TCR and MHC information</bold>
</td>
<td valign="top" align="left">932</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>Iedb experiment verified MHC ligand</bold>
</td>
<td valign="top" align="left">64,288</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>The number of predicted epitopes is restricted to different MHC I <bold>(A)</bold> and MHC II <bold>(B)</bold> alleles. The number of epitopes with a median binding affinity IC<sub>50</sub>&lt;500 nM from multiple affinity binding algorithms is labeled in red, and the number of epitopes with a median affinity IC<sub>50</sub> &#x2265;500 nM is labeled in blue.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394593-g002.tif"/>
</fig>
</sec>
<sec id="s4_2">
<title>Annotation of epitopes and mutations</title>
<p>After generation of predicted potential epitopes using our pipeline, the binding stability was annotated using antigen.garish (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>). Then, the expression levels of the corresponding genes carrying the mutations were annotated using TCGA data. A total of 64,288 epitopes have been experimentally verified and recorded in the IEDB (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>). A total of 542,442 mutations in MONET were recorded in the dbSNP database, which contains human variants including small indels original from germline or somatic mutations. Of these mutations, 13,289 are linked to the ClinVar database, which associates human variation with their potentially clinically relevant results. While most of the variant recorded in dbSNP and ClinVar are of germline origin, 198 somatic mutations in MONET have been recorded in ClinVar. The top ClinVar phenotypes (<xref ref-type="supplementary-material" rid="SM1">
<bold>Table S1</bold>
</xref>) include malignant tumor of prostate (Rank 1, n = 31), carcinoma of colon (Rank 2, n= 24), and colorectal cancer (Rank 5, n=16). In one hand, these clinical records verified our putative neoAgs that might present in dMMR/MSI-H cancers. In the other hand, the limited coverage of the current clinically available databases suggests that our computational results could be very valuable in bridging this gap.</p>
</sec>
<sec id="s4_3">
<title>Web interface</title>
<p>In MONET, the landing page can be accessed at <ext-link ext-link-type="uri" xlink:href="https://monet.mdanderson.org">https://monet.mdanderson.org</ext-link>. The key functions of MONET are: &#x2018;Search Neoantigens&#x2019; and &#x2018;Best Neoantigens&#x2019;. Both can be accessed from the sidebar.</p>
</sec>
<sec id="s4_4">
<title>Search neoantigens.</title>
<p>In the &#x2018;Search Neoantigens&#x2019; page, users have the option to search for neoAg using various parameters, including gene symbol and related IDs, epitope sequence, mutation HVGSp/HVGSc ID, and ClinVar phenotype. Moreover, users can narrow down their search by limiting it to specific MHC class I and II types. This functionality proves particularly useful when users have a specific target gene or disease in mind and would like to search into the details of a particular epitope (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>).</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Screenshots of MONET functions. <bold>(A)</bold> The screenshot shows the interface for the search function of neoAgs; <bold>(B)</bold> The screenshot shows the interface for identifying the best neoAgs restricted to a set of MHC allele combinations.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394593-g003.tif"/>
</fig>
<p>As an example, users can utilize the search function to find potential Class I neoAg epitopes resulting from mutations in <italic>TGFBR2</italic>. The search engine generates 520 summarized results, providing information such as peptide sequence, mutation location, related gene, and associated HLA (Human Leukocyte Antigen) alleles. To further refine the results, users can utilize the filter sidebar on the right-hand side. This allows filtering based on peptide sequence, affinity cutoff, and target HLA alleles. In cases where multiple genes are returned, users can also apply a filter based on the gene symbol. Once a specific epitope of interest is identified, users can select it to further access detailed information about the epitope (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>).</p>
<p>Detailed epitope information is offered in five tabs, each providing specific details (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>): 1. <italic>Prediction affinity</italic> will display a table of predicted affinities generated by different algorithms for various HLA types (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>) and other features of the epitope such as the Gravy score; 2. <italic>The mutant gene</italic> will show essential gene information related to the epitope, including the gene&#x2019;s location, related IDs, and both wild-type and mutant sequences of the protein. The mutant sequence is highlighted in red (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>); 3. <italic>ClinVar</italic> and <italic>dbSNP</italic> will provide access to the gene ID and corresponding link to ClinVar and dbSNP databases. Additionally, if available, basic information about related diseases and frequency data will be presented (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2</bold>
</xref>); 4. If the peptide has been experimentally tested in the <italic>IEDB</italic> (Immune Epitope Database), this tab will showcase relevant information from IEDB; 5. Users can explore the expression of the target gene in solid normal tissue versus primary solid tumors across different <italic>TCGA databases</italic>. Thus, this tab displays a bar plot illustrating this information (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S3</bold>
</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Screenshot of detailed reports of the epitope result page.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394593-g004.tif"/>
</fig>
</sec>
<sec id="s4_5">
<title>Best neoantigens</title>
<p>Within the best neoAg search engine function, users have the capability to search for all potential epitopes specific to one or multiple HLA alleles (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>). This feature allows for customized filtering options, such as an affinity and expression cutoff in a specific TCGA cancer type. This functionality proves particularly valuable when users intend to identify the top potential epitopes for a specific set of HLA alleles from either an individual patient or a group of patients.</p>
<p>As an example, users can select HLA-A*02:01 and HLA-B*07:02 alleles along with a median affinity cutoff of 30 nM across multiple algorithms. Furthermore, they can narrow down the search to include only the top 20% most highly expressed genes in the COAD dataset (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>). In total, 3,186 results for HLA-A*02:01 and 2,614 results for HLA-B*07:02 returned. These potential epitopes are generated from mutations in 1,483 genes. Users have the option to download the results directly from the page, and detailed information for each peptide can be accessed on the corresponding peptide details page.</p>
</sec>
<sec id="s4_6">
<title>Evaluating the coverage of neoantigens derived from microsatellite loci across various databases</title>
<p>The most distinctive feature of MONET is its focus on epitopes derived from microsatellite loci that are targets of MMRd. We compared the coverage of 65 of such epitopes, the immunogenicity of which has been validated using ELISpot assays (<xref ref-type="bibr" rid="B10">10</xref>), across MONET and other popular databases. MONET covers 57 out of these 65 verified epitopes (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>). In comparison to MONET, TSNAdb, only covers 15 of the 65 epitopes, and IEDB covers just one. GNIFdb, CAD, NEPdb, CAPAD, and AntiJen do not contain any of these epitopes. Therefore, MONET demonstrates excellent coverage (&gt;85%) of our target epitopes, which are derived from microsatellite loci due to MMRd, thus significantly outperforming other available databases in this field.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Coverage of validated epitopes derived from microsatellite loci across different databases. Y-axis: Number of validated epitopes; X-axis: Database name; Blue bars: Number of validated epitopes present in each database; Red bars: Number of validated epitopes missing from each database.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-15-1394593-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>The MONET database serves as a comprehensive resource for researchers investigating neoAg related to MSI cancers. This database covers possible neoAg derived from microsatellites within the human genome, which is specific to a panel of high-frequency MHC class I and class II alleles. Researchers can utilize the database to conduct searches based on target genes or epitope sequences, as well as target MHC alleles, to obtain comprehensive information on potential target epitopes. The user-friendly interface makes the database accessible to cancer researchers without acquiring any bioinformatic skills.</p>
<p>We acknowledge that MONET has several limitations. MONET solely focuses on target epitopes for humans, but we have ongoing efforts to broaden its scope and incorporate epitopes for other model systems such as mouse, rat, and rhesus, which will be particularly valuable for cancer vaccine studies using model organisms. MONET uses multiple MHC-peptide binding affinity algorithms to identify potential neoAgs. We currently treat all algorithms equally and calculate the minimum, median, or mean value of these algorithms to evaluate the peptides. However, these algorithms have large differences in performance (<xref ref-type="bibr" rid="B29">29</xref>, <xref ref-type="bibr" rid="B30">30</xref>). In the future, we plan to exclude those algorithms with inferior performance and prioritize the weight of algorithms with the best performance based on data generated by our team or public data sets. MONET currently lacks a REST API (Representational State Transfer Application Programming Interface), which could streamline data access and analysis for bioinformaticians seeking customized or bulk queries. A REST API is planned for integration in the upcoming MONET release.</p>
<p>In summary, MONET is a systems biology tool that has the goal of facilitating the identification of mutated neoAgs derived from microsatellite loci by leveraging publicly available state-of-the-art tools and by providing a user-friendly online website that is freely available to the scientific community.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>ND: Conceptualization, Data curation, Formal analysis, Methodology, Software, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. KS: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. EV: Formal analysis, Funding acquisition, Project administration, Resources, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by a gift from the Feinberg Family Foundation and grants R01CA260761, R01CA257375 and U01 CA233056 (US National Institutes of Health/National Cancer Institute) to EV; the generous philanthropic contributions to The University of Texas MD Anderson Cancer Center Moon Shots Program, The MD Anderson Cancer Center SPORE in Gastrointestinal Cancer P50 CA221707 (US National Institutes of Health/National Cancer Institute); and P30 CA016672 (US National Institutes of Health/National Cancer Institute) to the University of Texas MD Anderson Cancer Center Core Support Grant.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We thank Gaston Benavides Jr., Jenny Chen, and Rui Jiang for their IT support. We also thank Greg Holland and Ana Bolivar for their testing and suggestions on the webpage design.</p>
</ack>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest </title>
<p>EV has a consulting or advisory role with Janssen Research and Development, Recursion Pharma, Guardant Health, The Rising Tide Foundation, and Nouscom, s.r.l. EV has received research support from Janssen Research and Development.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fimmu.2024.1394593/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fimmu.2024.1394593/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<fn-group>
<title>Abbreviations</title>
<fn fn-type="abbr">
<p>MONET, The Microsatellite Neoantigen Discovery Tool; neoAg, Neoantigen; MHC, Major Histocompatibility Complex; MMR, Mismatch Repair; MSI, Microsatellite Instability.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vilar</surname> <given-names>E</given-names>
</name>
<name>
<surname>Gruber</surname> <given-names>SB</given-names>
</name>
</person-group>. <article-title>Microsatellite instability in colorectal cancer-the stable evidence</article-title>. <source>Nat Rev Clin Oncol</source>. (<year>2010</year>) <volume>7</volume>(<issue>3</issue>):<page-range>153&#x2013;62</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nrclinonc.2009.237</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schumacher</surname> <given-names>TN</given-names>
</name>
<name>
<surname>Scheper</surname> <given-names>W</given-names>
</name>
<name>
<surname>Kvistborg</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>Cancer neoantigens</article-title>. <source>Annu Rev Immunol</source>. (<year>2019</year>) <volume>37</volume>:<fpage>173</fpage>&#x2013;<lpage>200</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev-immunol-042617-053402</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lang</surname> <given-names>F</given-names>
</name>
<name>
<surname>Schrors</surname> <given-names>B</given-names>
</name>
<name>
<surname>Lower</surname> <given-names>M</given-names>
</name>
<name>
<surname>Tureci</surname> <given-names>O</given-names>
</name>
<name>
<surname>Sahin</surname> <given-names>U</given-names>
</name>
</person-group>. <article-title>Identification of neoantigens for individualized therapeutic cancer vaccines</article-title>. <source>Nat Rev Drug Discov</source>. (<year>2022</year>) <volume>21</volume>(<issue>4</issue>):<page-range>261&#x2013;82</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41573-021-00387-y</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vita</surname> <given-names>R</given-names>
</name>
<name>
<surname>Mahajan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Overton</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Dhanda</surname> <given-names>SK</given-names>
</name>
<name>
<surname>Martini</surname> <given-names>S</given-names>
</name>
<name>
<surname>Cantrell</surname> <given-names>JR</given-names>
</name>
<etal/>
</person-group>. <article-title>The immune epitope database (IEDB): 2018 update</article-title>. <source>Nucleic Acids Res</source>. (<year>2019</year>) <volume>47</volume>(<issue>D1</issue>):<page-range>D339&#x2013;43</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gky1006</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Toseland</surname> <given-names>CP</given-names>
</name>
<name>
<surname>Clayton</surname> <given-names>DJ</given-names>
</name>
<name>
<surname>McSparron</surname> <given-names>H</given-names>
</name>
<name>
<surname>Hemsley</surname> <given-names>SL</given-names>
</name>
<name>
<surname>Blythe</surname> <given-names>MJ</given-names>
</name>
<name>
<surname>Paine</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>AntiJen: a quantitative immunology database integrating functional, thermodynamic, kinetic, biophysical, and cellular data</article-title>. <source>Immunome Res</source>. (<year>2005</year>) <volume>1</volume>(<issue>1</issue>):<fpage>4</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1745-7580-1-4</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hutchison</surname> <given-names>S</given-names>
</name>
<name>
<surname>Pritchard</surname> <given-names>AL</given-names>
</name>
</person-group>. <article-title>Identifying neoantigens for use in immunotherapy</article-title>. <source>Mamm Genome</source>. (<year>2018</year>) <volume>29</volume>(<issue>11-12</issue>):<page-range>714&#x2013;30</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00335-018-9771-6</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>W</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>T</given-names>
</name>
<name>
<surname>Li</surname> <given-names>M</given-names>
</name>
<name>
<surname>He</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>GNIFdb: a neoantigen intrinsic feature database for glioma</article-title>. <source>Database (Oxford)</source>. (<year>2022</year>) <volume>2022</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/database/baac004</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>W</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Chi</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Hua</surname> <given-names>X</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>TSNAdb v2.0: the updated version of tumor-specific neoantigen database</article-title>. <source>Genomics Proteomics Bioinf</source>. (<year>2023</year>) <volume>21</volume>(<issue>2</issue>):<page-range>259&#x2013;66</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2022.07.28.501872</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>W</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>B</given-names>
</name>
<name>
<surname>Su</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Gu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Z</given-names>
</name>
<etal/>
</person-group>. <article-title>TSNAdb: A database for tumor-specific neoantigens from immunogenomics data analysis</article-title>. <source>Genomics Proteomics Bioinf</source>. (<year>2018</year>) <volume>16</volume>(<issue>4</issue>):<page-range>276&#x2013;82</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.gpb.2018.06.003</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bolivar</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Duzagac</surname> <given-names>F</given-names>
</name>
<name>
<surname>Deng</surname> <given-names>N</given-names>
</name>
<name>
<surname>Reyes-Uribe</surname> <given-names>L</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>K</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>W</given-names>
</name>
<etal/>
</person-group>. <article-title>Genomic landscape of lynch syndrome colorectal neoplasia identifies shared mutated neoantigens for immunoprevention</article-title>. <source>Gastroenterology</source>. (<year>2024</year>) <volume>166</volume>(<issue>5</issue>):<page-range>787&#x2013;801.e11</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1053/j.gastro.2024.01.016</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Niu</surname> <given-names>B</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>K</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>C</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>M</given-names>
</name>
<name>
<surname>McLellan</surname> <given-names>MD</given-names>
</name>
<etal/>
</person-group>. <article-title>MSIsensor: microsatellite instability detection using paired tumor-normal sequence data</article-title>. <source>Bioinformatics</source>. (<year>2014</year>) <volume>30</volume>(<issue>7</issue>):<page-range>1015&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btt755</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Greenbaum</surname> <given-names>J</given-names>
</name>
<name>
<surname>Sidney</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chung</surname> <given-names>J</given-names>
</name>
<name>
<surname>Brander</surname> <given-names>C</given-names>
</name>
<name>
<surname>Peters</surname> <given-names>B</given-names>
</name>
<name>
<surname>Sette</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Functional classification of class II human leukocyte antigen (HLA) molecules reveals seven different supertypes and a surprising degree of repertoire sharing across supertypes</article-title>. <source>Immunogenetics</source>. (<year>2011</year>) <volume>63</volume>(<issue>6</issue>):<page-range>325&#x2013;35</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00251-011-0513-0</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weiskopf</surname> <given-names>D</given-names>
</name>
<name>
<surname>Angelo</surname> <given-names>MA</given-names>
</name>
<name>
<surname>de Azeredo</surname> <given-names>EL</given-names>
</name>
<name>
<surname>Sidney</surname> <given-names>J</given-names>
</name>
<name>
<surname>Greenbaum</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Fernando</surname> <given-names>AN</given-names>
</name>
<etal/>
</person-group>. <article-title>Comprehensive analysis of dengue virus-specific responses supports an HLA-linked protective role for CD8+ T cells</article-title>. <source>Proc Natl Acad Sci USA</source>. (<year>2013</year>) <volume>110</volume>(<issue>22</issue>):<page-range>E2046&#x2013;53</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1305227110</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>O&#x2019;Donnell</surname> <given-names>TJ</given-names>
</name>
<name>
<surname>Rubinsteyn</surname> <given-names>A</given-names>
</name>
<name>
<surname>Laserson</surname> <given-names>U</given-names>
</name>
</person-group>. <article-title>MHCflurry 2.0: improved pan-allele prediction of MHC class I-presented peptides by incorporating antigen processing</article-title>. <source>Cell Syst</source>. (<year>2020</year>) <volume>11</volume>(<issue>4</issue>):<page-range>418&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cels.2020.09.001</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shao</surname> <given-names>XM</given-names>
</name>
<name>
<surname>Bhattacharya</surname> <given-names>R</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Sivakumar</surname> <given-names>IKA</given-names>
</name>
<name>
<surname>Tokheim</surname> <given-names>C</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>High-throughput prediction of MHC class I and II neoantigens with MHCnuggets</article-title>. <source>Cancer Immunol Res</source>. (<year>2020</year>) <volume>8</volume>(<issue>3</issue>):<fpage>396</fpage>&#x2013;<lpage>408</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/2326-6066.CIR-19-0464</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lundegaard</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lamberth</surname> <given-names>K</given-names>
</name>
<name>
<surname>Harndahl</surname> <given-names>M</given-names>
</name>
<name>
<surname>Buus</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lund</surname> <given-names>O</given-names>
</name>
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>NetMHC-3.0: accurate web accessible predictions of human, mouse and monkey MHC class I affinities for peptides of length 8-11</article-title>. <source>Nucleic Acids Res</source>. (<year>2008</year>) <volume>36</volume>:<page-range>W509&#x2013;12</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkn202</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Lund</surname> <given-names>O</given-names>
</name>
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>The PickPocket method for predicting binding specificities for receptors based on receptor pocket similarities: application to MHC-peptide binding</article-title>. <source>Bioinformatics</source>. (<year>2009</year>) <volume>25</volume>(<issue>10</issue>):<page-range>1293&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btp137</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lundegaard</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lund</surname> <given-names>O</given-names>
</name>
</person-group>. <article-title>Prediction of MHC class II binding affinity using SMM-align, a novel stabilization matrix alignment method</article-title>. <source>BMC Bioinf</source>. (<year>2007</year>) <volume>8</volume>:<fpage>238</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1471-2105-8-238</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Andreatta</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>NNAlign: a platform to construct and evaluate artificial neural network models of receptor-ligand interactions</article-title>. <source>Nucleic Acids Res</source>. (<year>2017</year>) <volume>45</volume>(<issue>W1</issue>):<page-range>W344&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkx276</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hundal</surname> <given-names>J</given-names>
</name>
<name>
<surname>Kiwala</surname> <given-names>S</given-names>
</name>
<name>
<surname>McMichael</surname> <given-names>J</given-names>
</name>
<name>
<surname>Miller</surname> <given-names>CA</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>H</given-names>
</name>
<name>
<surname>Wollam</surname> <given-names>AT</given-names>
</name>
<etal/>
</person-group>. <article-title>pVACtools: A computational toolkit to identify and visualize cancer neoantigens</article-title>. <source>Cancer Immunol Res</source>. (<year>2020</year>) <volume>8</volume>(<issue>3</issue>):<page-range>409&#x2013;20</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/2326-6066.CIR-19-0401</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jorgensen</surname> <given-names>KW</given-names>
</name>
<name>
<surname>Rasmussen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Buus</surname> <given-names>S</given-names>
</name>
<name>
<surname>Nielsen</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>NetMHCstab - predicting stability of peptide-MHC-I complexes; impacts for cytotoxic T lymphocyte epitope discovery</article-title>. <source>Immunology</source>. (<year>2014</year>) <volume>141</volume>(<issue>1</issue>):<fpage>18</fpage>&#x2013;<lpage>26</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/imm.12160</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Richman</surname> <given-names>LP</given-names>
</name>
<name>
<surname>Vonderheide</surname> <given-names>RH</given-names>
</name>
<name>
<surname>Rech</surname> <given-names>AJ</given-names>
</name>
</person-group>. <article-title>Neoantigen dissimilarity to the self-proteome predicts immunogenicity and response to immune checkpoint blockade</article-title>. <source>Cell Syst</source>. (<year>2019</year>) <volume>9</volume>(<issue>4</issue>):<fpage>375</fpage>&#x2013;<lpage>382.e4</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cels.2019.08.009</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McLaren</surname> <given-names>W</given-names>
</name>
<name>
<surname>Gil</surname> <given-names>L</given-names>
</name>
<name>
<surname>Hunt</surname> <given-names>SE</given-names>
</name>
<name>
<surname>Riat</surname> <given-names>HS</given-names>
</name>
<name>
<surname>Ritchie</surname> <given-names>GR</given-names>
</name>
<name>
<surname>Thormann</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>The ensembl variant effect predictor</article-title>. <source>Genome Biol</source>. (<year>2016</year>) <volume>17</volume>(<issue>1</issue>):<fpage>122</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13059-016-0974-4</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Westcott</surname> <given-names>PMK</given-names>
</name>
<name>
<surname>Sacks</surname> <given-names>NJ</given-names>
</name>
<name>
<surname>Schenkel</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Ely</surname> <given-names>ZA</given-names>
</name>
<name>
<surname>Smith</surname> <given-names>O</given-names>
</name>
<name>
<surname>Hauck</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Low neoantigen expression and poor T-cell priming underlie early immune escape in colorectal cancer</article-title>. <source>Nat Cancer</source>. (<year>2021</year>) <volume>2</volume>(<issue>10</issue>):<page-range>1071&#x2013;85</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s43018-021-00247-z</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Borden</surname> <given-names>ES</given-names>
</name>
<name>
<surname>Ghafoor</surname> <given-names>S</given-names>
</name>
<name>
<surname>Buetow</surname> <given-names>KH</given-names>
</name>
<name>
<surname>LaFleur</surname> <given-names>BJ</given-names>
</name>
<name>
<surname>Wilson</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Hastings</surname> <given-names>KT</given-names>
</name>
</person-group>. <article-title>NeoScore integrates characteristics of the neoantigen:MHC class I interaction and expression to accurately prioritize immunogenic neoantigens</article-title>. <source>J Immunol</source>. (<year>2022</year>) <volume>208</volume>(<issue>7</issue>):<page-range>1813&#x2013;27</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4049/jimmunol.2100700</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wells</surname> <given-names>DK</given-names>
</name>
<name>
<surname>van Buuren</surname> <given-names>MM</given-names>
</name>
<name>
<surname>Dang</surname> <given-names>KK</given-names>
</name>
<name>
<surname>Hubbard-Lucey</surname> <given-names>VM</given-names>
</name>
<name>
<surname>Sheehan</surname> <given-names>KCF</given-names>
</name>
<name>
<surname>Campbell</surname> <given-names>KM</given-names>
</name>
<etal/>
</person-group>. <article-title>Key parameters of tumor epitope immunogenicity revealed through a consortium approach improve neoantigen prediction</article-title>. <source>Cell</source>. (<year>2020</year>) <volume>183</volume>(<issue>3</issue>):<fpage>818</fpage>&#x2013;<lpage>34.e13</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cell.2020.09.015</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sherry</surname> <given-names>ST</given-names>
</name>
<name>
<surname>Ward</surname> <given-names>MH</given-names>
</name>
<name>
<surname>Kholodov</surname> <given-names>M</given-names>
</name>
<name>
<surname>Baker</surname> <given-names>J</given-names>
</name>
<name>
<surname>Phan</surname> <given-names>L</given-names>
</name>
<name>
<surname>Smigielski</surname> <given-names>EM</given-names>
</name>
<etal/>
</person-group>. <article-title>dbSNP: the NCBI database of genetic variation</article-title>. <source>Nucleic Acids Res</source>. (<year>2001</year>) <volume>29</volume>(<issue>1</issue>):<page-range>308&#x2013;11</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/29.1.308</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Landrum</surname> <given-names>MJ</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Benson</surname> <given-names>M</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>GR</given-names>
</name>
<name>
<surname>Chao</surname> <given-names>C</given-names>
</name>
<name>
<surname>Chitipiralla</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>ClinVar: improving access to variant interpretations and supporting evidence</article-title>. <source>Nucleic Acids Res</source>. (<year>2018</year>) <volume>46</volume>(<issue>D1</issue>):<page-range>D1062&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkx1153</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bonsack</surname> <given-names>M</given-names>
</name>
<name>
<surname>Hoppe</surname> <given-names>S</given-names>
</name>
<name>
<surname>Winter</surname> <given-names>J</given-names>
</name>
<name>
<surname>Tichy</surname> <given-names>D</given-names>
</name>
<name>
<surname>Zeller</surname> <given-names>C</given-names>
</name>
<name>
<surname>Kupper</surname> <given-names>MD</given-names>
</name>
<etal/>
</person-group>. <article-title>Performance evaluation of MHC class-I binding prediction tools based on an experimentally validated MHC-peptide binding data set</article-title>. <source>Cancer Immunol Res</source>. (<year>2019</year>) <volume>7</volume>(<issue>5</issue>):<page-range>719&#x2013;36</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1158/2326-6066.CIR-18-0584</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mei</surname> <given-names>S</given-names>
</name>
<name>
<surname>Li</surname> <given-names>F</given-names>
</name>
<name>
<surname>Leier</surname> <given-names>A</given-names>
</name>
<name>
<surname>Marquez-Lago</surname> <given-names>TT</given-names>
</name>
<name>
<surname>Giam</surname> <given-names>K</given-names>
</name>
<name>
<surname>Croft</surname> <given-names>NP</given-names>
</name>
<etal/>
</person-group>. <article-title>A comprehensive review and performance evaluation of bioinformatics tools for HLA class I peptide-binding prediction</article-title>. <source>Brief Bioinform</source>. (<year>2020</year>) <volume>21</volume>(<issue>4</issue>):<page-range>1119&#x2013;35</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bib/bbz051</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>