<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Immunol.</journal-id>
<journal-title>Frontiers in Immunology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Immunol.</abbrev-journal-title>
<issn pub-type="epub">1664-3224</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fimmu.2025.1528524</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Immunology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Serologic biomarker discovery for differentiating Lyme disease from diseases with similar clinical symptoms using broad profiling of antibody binding</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Tingting</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3054654/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Baert</surname>
<given-names>Laurie</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2344082/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Woodbury</surname>
<given-names>Neal W.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2611544/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kelbauskas</surname>
<given-names>Laimonas</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/123265/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Biodesign Institute, Arizona State University</institution>, <addr-line>Tempe, AZ</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Immunology, Mayo Clinic</institution>, <addr-line>Scottsdale, AZ</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Biomorph Technologies</institution>, <addr-line>Chandler, AZ</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Caian Vinhaes, Multinational Organization Network Sponsoring Translational and Epidemiological Research (MONSTER), Brazil</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Lenny Moise, SeromYx Systems, United States</p>
<p>John Shearer Lambert, University College Dublin, Ireland</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Laimonas Kelbauskas, <email xlink:href="mailto:lkelbaus@asu.edu">lkelbaus@asu.edu</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>05</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1528524</elocation-id>
<history>
<date date-type="received">
<day>15</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>04</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Zhang, Baert, Woodbury and Kelbauskas</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zhang, Baert, Woodbury and Kelbauskas</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Lyme disease (LD) is a tick-borne disease that is a substantial public health burden with estimated about 0.5 million new cases per year in the US and increasing incidence. Differentiating Lyme disease, especially in its early stages, from other febrile illnesses with similar clinical symptoms (look-alike diseases) represents a significant challenge due to the lack of diagnostic tools. Current diagnostic tools based on serology were not specifically developed for differential diagnosis and show limited sensitivity in early LD resulting in high false negative rates.</p>
</sec>
<sec>
<title>Methods</title>
<p>The work presented here focuses on a broad profiling of the humoral immune response in terms of circulating antibody repertoire in patients diagnosed with LD and a number of diseases with similar clinical symptoms. A combination of antibody binding to a library of linear, diverse peptides and machine learning methods revealed a panel of biomarker proteins from the proteome of the Borrelia burgdorferi bacterium (LD causing pathogen) that can be used to differentiate between LD and other diseases.</p>
</sec>
<sec>
<title>Results</title>
<p>A subset of the biomarkers was independently validated and demonstrated to show robust differentiating power. Importantly, the discovered biomarkers distinguish between LD patients that previously tested negative with the current test standard (false negatives) and the look-alike diseases.</p>
</sec>
<sec>
<title>Discussion</title>
<p>These findings are important in that the discovered biomarkers can be utilized for differential diagnosis of LD. Furthermore, because the discovery approach is agnostic, the results suggest that it can also be used for biomarker discovery of other diseases.</p>
</sec>
</abstract>
<kwd-group>
<kwd>humoral immune response</kwd>
<kwd>Lyme disease</kwd>
<kwd>differentiating diagnosis</kwd>
<kwd>machine learning</kwd>
<kwd>peptide array analyses</kwd>
<kwd>seronegative Lyme disease</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="4"/>
<equation-count count="4"/>
<ref-count count="48"/>
<page-count count="16"/>
<word-count count="9721"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Systems Immunology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>Tick-borne diseases (TBDs) have become a major public health challenge with a projected incidence rate of &gt;35% of the global population by 2050 (<xref ref-type="bibr" rid="B1">1</xref>). Lyme disease (LD) is the most prevalent tick-borne zoonotic disease in the USA with an estimated 476,000 new cases each year and increasing incidence (<xref ref-type="bibr" rid="B2">2</xref>). Despite recent advances, the diagnosis of LD, especially in the early stages of the disease, remains challenging due to the limitations of the current testing methods. The current standard for LD diagnosis is based on the detection of antibodies raised by the humoral arm of the human immune system against specific antigens from the <italic>Borrelia burgdorferi</italic> proteome. These tests are only ~30% sensitive in the early stages of the disease at 96% specificity (<xref ref-type="bibr" rid="B3">3</xref>). Adding to the diagnostic challenge is the fact that LD typically presents with symptoms such as fever, muscle pain, and fatigue, which are shared with a number of other common diseases like the flu or seasonal cold. The &#x201c;bullseye rash&#x201d; [erythema migrans (EM)], a typical tell-tale sign of LD, does not appear in a substantial portion of patients or presents with differing morphology, further complicating diagnosis (<xref ref-type="bibr" rid="B4">4</xref>&#x2013;<xref ref-type="bibr" rid="B6">6</xref>). Furthermore, there is increasing evidence that an EM-like rash can be present in patients after exposure to pathogens other than members of the <italic>B. burgdorferi sensu lato</italic> complex (<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B8">8</xref>). As a result, misdiagnosis with another disease can result in poor treatment outcomes.</p>
<p>A <italic>B. burgdorferi</italic> infection can involve a range of organs resulting in dermatological, cardiac, neurological, and musculoskeletal disorders. Successful differential diagnosis of Lyme disease against diseases with look-alike symptoms is required for timely treatment when antibiotics are most effective (<xref ref-type="bibr" rid="B9">9</xref>). Delays in diagnosis in approximately 40% of patients result from an absence of an EM rash, unnoticed tick bite, human factors, and confounding symptoms that indicate another disease (<xref ref-type="bibr" rid="B10">10</xref>). Other TBDs, such as <italic>Babesia microti</italic> and <italic>Ehrlichia</italic> spp., are spread by the same tick as <italic>B. burgdorferi</italic> and result in febrile illness with similar symptoms (<xref ref-type="bibr" rid="B11">11</xref>). One problem with the current diagnostic is that it was developed to differentiate LD from healthy controls. However, due to the significant symptom overlap with other, common diseases, the development of a diagnostic approach designed to distinguish between LD and a broad range of diseases would be very beneficial.</p>
<p>Lyme disease can be misdiagnosed as influenza, Epstein&#x2013;Barr virus (EBV), and parvovirus B19, which cause similar fever, myalgias, and fatigue (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B13">13</xref>). Cross-reactivity of antibodies raised during EBV and syphilis and against autoimmune markers on current serologic tests further confounds diagnostic tests (<xref ref-type="bibr" rid="B14">14</xref>&#x2013;<xref ref-type="bibr" rid="B17">17</xref>).</p>
<p>
<italic>B. burgdorferi</italic> presents unique challenges in even the acute disease presentation that are associated with its innate ability to modulate host immune system response (<xref ref-type="bibr" rid="B18">18</xref>). It is a highly antigenically heterogenic genospecies with 25 known serotypes of Outer Surface Protein C (OspC) alone (<xref ref-type="bibr" rid="B19">19</xref>). In addition, different strains carry different combinations of extra-genomic plasmids that encode immunogenic antigens (<xref ref-type="bibr" rid="B20">20</xref>, <xref ref-type="bibr" rid="B21">21</xref>); recombinant antigenic variation alters the VlsE surface protein, which enhances immune evasion (<xref ref-type="bibr" rid="B22">22</xref>&#x2013;<xref ref-type="bibr" rid="B24">24</xref>); there is geographic variation in antigens recognized by serum antibodies (<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B26">26</xref>); and co-infection can occur with multiple <italic>B. burgdorferi</italic> strains (<xref ref-type="bibr" rid="B27">27</xref>, <xref ref-type="bibr" rid="B28">28</xref>) or with other tick-borne diseases such as <italic>Babesia</italic> (<xref ref-type="bibr" rid="B28">28</xref>, <xref ref-type="bibr" rid="B29">29</xref>). Within the <italic>B. burgdorferi sensu lato</italic> complex, five genospecies including <italic>B. burgdorferi</italic>, <italic>Borrelia afzelii</italic>, and <italic>Borrelia garinii</italic> (with the latter two mainly found in Europe) are known to cause LD in humans (<xref ref-type="bibr" rid="B30">30</xref>), while others such as <italic>Borrelia mayonii</italic> cause an LD-like illness, and <italic>Borrelia miyamotoi</italic>, a relapsing fever spirochete, circulates via the same tick (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B32">32</xref>). Direct detection methods targeting <italic>B. burgdorferi</italic> are limited due to the low bacterial load in the blood after the initial infection (<xref ref-type="bibr" rid="B33">33</xref>). Therefore, serology has been the method of choice for LD diagnosis based on the presence of antibodies specific to targets in the <italic>B. burgdorferi</italic> proteome.</p>
<p>This study was designed to address the question of whether a broad, agnostic profiling of the humoral immune response can identify a set of immunogenic proteins from the <italic>B. burgdorferi</italic> proteome that gives rise to antibodies with differential reactivity between LD and a number of other febrile diseases. Because LD is known to be highly heterogeneous in terms of the adaptive immune response of the host, as well as timing for disease progression and symptom severity, we utilized a method for broad and unbiased profiling of the circulating antibody binding repertoire in sera obtained from LD patients and from individuals diagnosed with other diseases that have similar clinical symptoms [look-alike diseases (LADs)].</p>
<p>To accomplish this, a random and sparse sampling of the entire combinatorial sequence space of short (6&#x2013;13 amino acids long) linear peptides was represented on a peptide array consisting of 126,051 unique, randomly designed peptides that do not represent any specific antigen or pathogen. After exposing the peptides to antibodies contained in a serum sample, the binding profile of each patient&#x2019;s antibodies to the peptide library is measured using a fluorescently labeled secondary polyclonal anti-IgG antibody, and this is read out as fluorescence intensity. A comprehensive sequence-binding relationship between the peptide array sequences and the measured IgG binding values is developed by training a machine learning (ML) algorithm. This model is then used to predict the total IgG binding in each serum sample to the proteins that make up the <italic>B. burgdorferi</italic> proteome or to the proteomes of other pathogens.</p>
<p>A number of candidate proteins from the <italic>B. burgdorferi</italic> proteome with predicted differential Ab reactivity are selected based on the predicted binding values. After selection, the candidate proteins were expressed in <italic>Escherichia coli</italic>, and their ability to differentiate between LD and look-alike diseases was evaluated on an orthogonal, bead-based assay.</p>
<p>Previous work from this lab (<xref ref-type="bibr" rid="B34">34</xref>&#x2013;<xref ref-type="bibr" rid="B41">41</xref>) has demonstrated the utility of this approach to identify linear epitopes of a number of monoclonal antibodies (<xref ref-type="bibr" rid="B42">42</xref>), differentiate among different infectious diseases (<xref ref-type="bibr" rid="B43">43</xref>), and reveal substantial person-to-person variability in humoral immune response in LD patients (<xref ref-type="bibr" rid="B36">36</xref>). These studies have shown that despite the inherent limitation of the linear peptides in identifying conformational (discontinuous) epitopes, the method is capable of providing biologically relevant insight into humoral immune responses by broadly characterizing antibody binding profiles in a disease-agnostic manner. Importantly, the investigation of LD humoral immune response also revealed a differential humoral response between the seropositive and seronegative LD cohorts, suggesting that the two patient subgroups should be considered distinct. The same study also reported strong similarity in antibody binding profiles between LD and healthy controls from geographies endemic to LD, but not other locations, indicating potentially high seroprevalence in areas with high LD incidence.</p>
<p>In comparison with the previous work, the current study presented here is based on a substantially expanded LD+ cohort as well as expanded look-alike disease cohorts to increase the statistical power and generalizability of the approach in identifying immunogenic targets in LD. In principle, this type of approach opens the door to the discovery of novel, potentially more potent biomarkers for disease diagnosis and provides a disease-agnostic means for the identification of new therapeutic targets for a number of infectious and autoimmune diseases for which currently no cure exists.</p>
<p>Here, the approach described above was used to select candidate biomarker proteins from the <italic>B. burgdorferi</italic> proteome that show predicted differentiation between LD patients and patients with a range of other febrile diseases that have similar symptoms, resulting in substantial improvement in the sensitivity of the assay over the current serologic test standard. This involved antibody profiling on peptide arrays, predicted binding to the entire <italic>B. burgdorferi</italic> proteome, and a subsequent candidate biomarker selection process, followed by a validation of the selected targets.</p>
</sec>
<sec id="s2" sec-type="results">
<title>Results</title>
<sec id="s2_1">
<title>Antibody binding profiles measured on peptide arrays</title>
<p>Antibody binding of a total of 536 human serum samples representing seropositive (LD+) and seronegative (LD&#x2212;) Lyme disease and 13 diseases with symptoms similar to LD (LADs; see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> for cohort breakdown) were profiled on the peptide arrays. The &#x201c;seropositive&#x201d; designation was assigned to the LD patients who presented with a rash of &gt;5-cm diameter and tested positive with the Centers for Disease Control and Prevention (CDC)-recommended serologic test. Lyme disease patients who presented with a rash of &gt;5-cm diameter but tested negative with standard two-tier test (STTT) were categorized as &#x201c;seronegative&#x201d; LD.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Discovery cohort breakdown.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Disease</th>
<th valign="top" align="left">Number of samples</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Seropositive LD (LD+)</td>
<td valign="top" align="left">186</td>
</tr>
<tr>
<td valign="top" align="left">Seronegative LD (LD&#x2212;)</td>
<td valign="top" align="left">102</td>
</tr>
<tr>
<td valign="top" align="left">Alcoholic liver disease</td>
<td valign="top" align="left">9</td>
</tr>
<tr>
<td valign="top" align="left">Antinuclear antibodies</td>
<td valign="top" align="left">15</td>
</tr>
<tr>
<td valign="top" align="left">
<italic>Babesia</italic>
</td>
<td valign="top" align="left">23</td>
</tr>
<tr>
<td valign="top" align="left">Chlamydia</td>
<td valign="top" align="left">12</td>
</tr>
<tr>
<td valign="top" align="left">Dengue</td>
<td valign="top" align="left">3</td>
</tr>
<tr>
<td valign="top" align="left">Epstein&#x2013;Barr virus</td>
<td valign="top" align="left">82</td>
</tr>
<tr>
<td valign="top" align="left">Fibromyalgia</td>
<td valign="top" align="left">1</td>
</tr>
<tr>
<td valign="top" align="left">Influenza</td>
<td valign="top" align="left">27</td>
</tr>
<tr>
<td valign="top" align="left">Mononucleosis</td>
<td valign="top" align="left">2</td>
</tr>
<tr>
<td valign="top" align="left">Parvovirus</td>
<td valign="top" align="left">9</td>
</tr>
<tr>
<td valign="top" align="left">Rheumatoid arthritis</td>
<td valign="top" align="left">14</td>
</tr>
<tr>
<td valign="top" align="left">Syphilis</td>
<td valign="top" align="left">19</td>
</tr>
<tr>
<td valign="top" align="left">West Nile virus</td>
<td valign="top" align="left">17</td>
</tr>
<tr>
<td valign="top" align="left">Total</td>
<td valign="top" align="left">521</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Anti-SSA, anti-Sjogren&#x2019;s syndrome-related antigen A autoantibodies; LD, Lyme disease.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The sequences of the peptides on the array were designed randomly with a goal of equally representing 19 (cysteine was excluded due to synthesis constraints) canonical amino acids on the array. As a result, the peptide arrays can be understood as an unbiased and agnostic way to evenly interrogate the binding of the patient&#x2019;s antibody repertoire to the entire combinatorial space of 10-mer peptides. Due to the random and agnostic peptide array design, any differences observed in antibody binding between the study cohorts can be attributed to a specific humoral immune system response and should not be affected by the peptide array design per se.</p>
<p>The distribution of binding intensities of total serum IgG to peptide array sequences shows an overall stronger binding for samples from the patient cohort with LADs compared to the LD+ and LD&#x2212; disease cohorts (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>). All three cohorts show bimodal distributions with two distinct peaks in their binding intensity. The first peak is centered at the low end of the binding intensity range close to the background binding signal on the array. Thus, the first peak likely represents the fraction of peptides that bind antibodies non-specifically and only weakly or not at all. The second peak in all three cohorts involves stronger binding peptide sequences. Despite the similar shape of the distributions between the three cohorts, the distributions show differences. The shift of the second peak with respect to the first varies markedly with cohort. The look-alike cohort shows the largest separation between the peaks with approximately 4&#xd7; higher intensity of the second peak (0.6 on the log10 scale). The seronegative LD (LD&#x2212;) group of patients exhibits the smallest ratio of ~1.6&#xd7; in the shift between the two peaks, with the seropositive (LD+) cohort showing an intermediate shift of ~2&#xd7;. The position of the second peak presumably captures the more specific binding and thus likely contains the disease-specific response. This suggests that antibodies in the look-alike disease group show an overall stronger immune response compared to both LD cohorts. Consistent with this, the LD+ cohort where clear antibody reactivity has been measured results in overall stronger binding than the LD&#x2212; cohort, presumably reflecting the larger number of reactive antibodies present (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Comparison of antibody binding to library peptides between the three cohorts. <bold>(A)</bold> Intensity distributions of binding to the peptide library on the microarray. The intensity values have been log-transformed to make them more normal-like. The Y axis represents density of counts. <bold>(B)</bold> UMAP representation of the data shown in panel A with look-alike diseases, LD&#x2212;, and LD+ represented as green, blue, and red circles, respectively. The dashed ovals represent three clusters within the distribution. UMAP, Uniform Manifold Approximation and Projection; LD, Lyme disease.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-16-1528524-g001.tif"/>
</fig>
<p>The binding intensities between the three cohorts were qualitatively compared using the Uniform Manifold Approximation and Projection (UMAP) method for data dimensionality reduction and visualization (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>). The three disease cohorts show substantial overlap with one another in this representation (also see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;1</bold>
</xref> for pairwise comparison using UMAP; the look-alike and LD&#x2212; cohorts show partial separation in a binary comparison; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;1B</bold>
</xref>). This suggests that despite the observed differences in the binding intensity distributions between the cohorts, differentiation based on the binding intensities of the individual peptide is not as clear. Interestingly, the UMAP representation suggests the presence of at least three subclusters in the data (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>), indicating underlying additional complexity with differences in fractional distributions among the cohorts. For example, cluster 3 contained only 27 out of 235 (11%) of the LAD patients, with the rest of the cohort distributed approximately evenly between clusters 1 and 2. In contrast, 39 out of 93 (42%) LD&#x2212; individuals were included in cluster 3, while 17 (18%) and 37 (40%) were contained in clusters 1 and 2, respectively. The LD+ patients are distributed approximately equally among the three clusters.</p>
<p>Because of the differences observed between the seropositive and seronegative LD groups compared with the look-alike diseases, the combined LD (LD+/LD&#x2212;) cohorts were compared with the look-alike disease cohort in determining p-value distributions and building classifier models. As observed earlier when analyzing the binding intensity distributions (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>), the majority of the peptides showed lower binding intensities in the LD cohorts compared with the look-alike diseases regardless of whether the LD cohorts were combined or not (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2A&#x2013;C</bold>
</xref>). A comparison of classification performance using the Extreme Gradient Boosting (XGBoost) algorithm (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2D&#x2013;F</bold>
</xref>) showed higher Area under the curve (AUC) values for both LD+ vs. LADs and LD&#x2212; vs. LADs (AUC = 0.83, 95% CI: 0.77&#x2013;0.98, and AUC = 0.85, 95% CI: 0.81&#x2013;0.89, respectively) compared to the combined (LD+/LD&#x2212;) vs. LADs [AUC = 0.77 (0.72&#x2013;0.82)]. This result indicates a somewhat better classification in terms of AUC value when the two LD cohorts are considered separately. The classification results imply that the LD+ and LD&#x2212; cohorts differ not only in terms of known biomarkers being positive in LD+ but also in terms of antibody reactivity that is unique to LD&#x2212;.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Separating LD+ and LD&#x2212; cohorts provides better classification performance compared to the combined LD+/LD&#x2212; cohort from the look-alike diseases. p-Value distributions (volcano plots; <bold>(A&#x2013;C)</bold>) and classification performance in terms of receiver operating characteristic (ROC) curves <bold>(D&#x2013;F)</bold> of the following contrasts between the cohorts: combined LD+/LD&#x2212; vs. look-alike diseases <bold>(A, D)</bold>, LD+ vs. look-alike diseases <bold>(B, E)</bold>, and LD&#x2212; vs. look-alike diseases <bold>(C, F)</bold>. In all three cases, an XGBoost classifier was trained on 90% randomly chosen peptides in the entire library (n = 126,051 peptides). The remaining 10% of the data were used for cross-validation, which was performed 10 times. The AUC values and their 95% confidence intervals are shown in the graphs. LD, Lyme disease; XGBoost, Extreme Gradient Boosting.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-16-1528524-g002.tif"/>
</fig>
</sec>
<sec id="s2_2">
<title>Predicted antibody binding to the <italic>B. burgdorferi</italic> proteome</title>
<p>Due to the random nature of the peptide sequences in the array library, one cannot directly use the information about the sequence-binding relationship to determine what proteins from the <italic>B. burgdorferi</italic> proteome may be immunogenic and serve as potential candidate biomarkers. To select proteins from the <italic>B. burgdorferi</italic> proteome for further validation, ML approaches were used to model the underlying sequence-binding patterns and then project the patterns onto all <italic>B. burgdorferi</italic> proteins. The sequences of each protein from the <italic>B. burgdorferi</italic> B31 strain proteome (n = 1,219) were split into 10 AA-long tiles with nine amino acids (AA) overlapping between adjacent tiles. The proteome tiles were then one-hot encoded and used as input for the neural network (NN) models trained on the peptide array binding data of each sample, resulting in predicted binding intensities of every tile. These were assembled into a predicted binding map of each protein. Note that one separate model was trained on the binding data from each patient, resulting in <italic>B. burgdorferi</italic> binding predictions generated for each patient. A comparison of the predicted binding distributions (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>) shows similar characteristics to the measured binding on the peptide array. All three cohorts show distinct bimodal shapes with the first peak representing weak binders and the second peak capturing mainly the stronger interactions. With respect to the LD+ and LD&#x2212; cohorts, the LAD group shows the largest separation between the two peaks. It is also shifted most toward the higher intensities (stronger interactions) compared with the other two cohorts, followed by the LD+ and LD&#x2212; groups. However, there is one substantial difference between the measured and predicted distributions. The second peak (strong interactions) in the distribution of predicted values to the <italic>B. burgdorferi</italic> proteome is higher than the first peak (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>), in contrast to that observed in the measured array peptide binding values (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>). This suggests that the sequences from the <italic>B. burgdorferi</italic> proteome overall contain more peptides (tiles) that resemble antigenic targets of antibodies in each patient. Interestingly, a UMAP representation of the predictions (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>) revealed a distribution with fewer distinct subclusters as compared to the measured intensities (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>) even though the shape and overall overlap between the distributions of the three cohorts are similar. In addition, the LAD and LD&#x2212; cohorts (green and blue dots in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>) appear to be separated more than observed in the measured data (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>), at least visually. Nevertheless, the overall similarity between the measured and predicted distributions indicates reliable model performance with regard to projecting patterns in the learned data onto a biologically relevant context.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Mapping peptide array binding data onto the <italic>Borrelia burgdorferi</italic> proteome. <bold>(A)</bold> Predicted binding intensity distributions. <bold>(B)</bold> UMAP representation of the predicted Ab binding intensities to the tiled <italic>B</italic>. <italic>burgdorferi</italic> proteome with look-alike diseases, LD&#x2212;, and LD+ represented as green, blue, and red circles, respectively. The distributions and the UMAP representation of the predictions show close similarity to the measured data. LD, Lyme disease; XGBoost, Extreme Gradient Boosting; UMAP, Uniform Manifold Approximation and Projection.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-16-1528524-g003.tif"/>
</fig>
</sec>
<sec id="s2_3">
<title>Statistical significance and classification performance using binding prediction to the <italic>B. burgdorferi</italic> proteome</title>
<p>The model predictions of antibody binding to the <italic>B. burgdorferi</italic> proteome were analyzed in terms of p-value distributions (<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4A&#x2013;C</bold>
</xref>). Similar distribution characteristics were observed to those of the data measured on the peptide arrays (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>), with the majority of the proteome tiles showing lower predicted binding intensities to antibodies in sera of the LD+ and LD&#x2212; cohorts compared to the LAD group of patients. Two arbitrary threshold ratio values of 0.8 and 1.2 were used to better highlight trends in the differential binding profiles. There are relatively few tiles that show stronger binding in the LD cohort than in the LAD cohort, consistent with the binding distributions for the cohorts (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>). Classifier models were trained on the predicted <italic>B. burgdorferi</italic> binding values (<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4D&#x2013;F</bold>
</xref>). Lower classification performance was observed compared to models trained on the measured peptide array data (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2D&#x2013;F</bold>
</xref>). All three classifiers showed similar AUC values, suggesting, in contrast to the results obtained with the peptide array data, comparable differentiation between the two LD and the LAD cohorts. Unlike the models trained on the measured peptide array data, the classification of the combined LD+/LD&#x2212; vs. LAD cohorts was similar to the classification using the separate cohorts.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Statistical significance of predicted antibody binding intensities to tiles of the <italic>Borrelia burgdorferi</italic> proteome <bold>(A&#x2013;C)</bold> and classification performance between the corresponding cohorts <bold>(D&#x2013;F)</bold>. The horizontal axis in <bold>(A&#x2013;C)</bold> represents the ratio of intensity means between the corresponding LD cohort and the LAD. The p-values were calculated using a t-test and are not adjusted for multiple hypothesis comparison to better highlight differences among the comparisons. The blue dots in <bold>(A&#x2013;C)</bold> depict tiles with predicted binding intensities below a threshold intensity ratio of 0.8 between the LD and LAD cohorts. LD, Lyme disease; LAD, look-alike disease.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-16-1528524-g004.tif"/>
</fig>
</sec>
<sec id="s2_4">
<title>Candidate protein biomarker selection from the <italic>B. burgdorferi</italic> proteome</title>
<p>Note that the analysis of predicted total IgG binding to the <italic>B. burgdorferi</italic> proteome, as described above, was performed at the level of separate sequence tiles and not complete proteins. This resulted in generally low effect sizes and mediocre classification performance, suggesting that there are no strong candidate biomarkers at the individual sequence tile level.</p>
<p>It is not very surprising that short sequence pieces of potential antigens do not provide strong differentiation. Past work has shown that the immune response to Lyme disease is very heterogeneous (<xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B44">44</xref>), and thus, even if patients have antibodies against common proteins, they may well not bind to the same epitopes in those proteins. An alternative is to consider the aggregate predicted binding to each protein in the proteome and in this way select candidate protein biomarkers from the <italic>B. burgdorferi</italic> proteome with high predicted differentiating power between the combined LD+ and LD&#x2212; cohorts and the LAD group of patients. Two different complementary methods were used for candidate protein selection. The first method was based on ranking the proteins by the lowest p-value determined from the predicted binding values of the tiles that make up its sequence. The p-values were calculated using the outlier sum statistics (<xref ref-type="bibr" rid="B45">45</xref>) (see Materials and Methods) selecting the proteins with a false discovery rate (FDR) of &lt;0.05. The outlier sum statistics method was chosen to account for the long tails in the binding intensity distributions observed on the peptide array assays. It has been demonstrated that outlier sum statistics outperforms the t-test method for calculating the statistical significance of distributions with outliers (<xref ref-type="bibr" rid="B45">45</xref>). This method of selection resulted in a total of 19 protein candidates being selected from a total of 1,281 proteins (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Data Sheet 1</bold></xref>) contained in the <italic>B. burgdorferi</italic> proteome (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>) (<xref ref-type="disp-formula" rid="eq1">Equations 1</xref>&#x2013;<xref ref-type="disp-formula" rid="eq4">4</xref>). The last two proteins in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> were included because the FDR values were above the 0.05 cutoff by only a small margin.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Selected candidate biomarker proteins from the <italic>Borrelia burgdorferi</italic> proteome using the outlier sum statistics method.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">UniProt ID</th>
<th valign="top" align="left">p-Value</th>
<th valign="top" align="left">FDR</th>
<th valign="top" align="left">Protein</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">O51353</td>
<td valign="top" align="left">2.92E&#x2212;06</td>
<td valign="top" align="left">3.69E&#x2212;03</td>
<td valign="top" align="left">50S ribosomal protein L1</td>
</tr>
<tr>
<td valign="top" align="left">O51555</td>
<td valign="top" align="left">9.42E&#x2212;06</td>
<td valign="top" align="left">5.60E&#x2212;03</td>
<td valign="top" align="left">Trigger factor</td>
</tr>
<tr>
<td valign="top" align="left">O51247</td>
<td valign="top" align="left">1.33E&#x2212;05</td>
<td valign="top" align="left">5.60E&#x2212;03</td>
<td valign="top" align="left">50S ribosomal protein L31 type B</td>
</tr>
<tr>
<td valign="top" align="left">P52323</td>
<td valign="top" align="left">2.13E&#x2212;05</td>
<td valign="top" align="left">6.34E&#x2212;03</td>
<td valign="top" align="left">RNA polymerase sigma factor RpoD</td>
</tr>
<tr>
<td valign="top" align="left">O51757</td>
<td valign="top" align="left">2.51E&#x2212;05</td>
<td valign="top" align="left">6.34E&#x2212;03</td>
<td valign="top" align="left">UDP-<italic>N</italic>-acetylmuramate&#x2013;<sc>l</sc>-alanine ligase</td>
</tr>
<tr>
<td valign="top" align="left">O51112</td>
<td valign="top" align="left">3.92E&#x2212;05</td>
<td valign="top" align="left">8.27E&#x2212;03</td>
<td valign="top" align="left">Uncharacterized protein BB_0085</td>
</tr>
<tr>
<td valign="top" align="left">O50893</td>
<td valign="top" align="left">1.45E&#x2212;04</td>
<td valign="top" align="left">2.42E&#x2212;02</td>
<td valign="top" align="left">HTH_OrfB_IS605 domain-containing protein</td>
</tr>
<tr>
<td valign="top" align="left">O51178</td>
<td valign="top" align="left">1.68E&#x2212;04</td>
<td valign="top" align="left">2.42E&#x2212;02</td>
<td valign="top" align="left">Uncharacterized protein</td>
</tr>
<tr>
<td valign="top" align="left">O51143</td>
<td valign="top" align="left">1.94E&#x2212;04</td>
<td valign="top" align="left">2.42E&#x2212;02</td>
<td valign="top" align="left">Pts system, maltose and glucose-specific IIABC component</td>
</tr>
<tr>
<td valign="top" align="left">O51632</td>
<td valign="top" align="left">1.77E&#x2212;04</td>
<td valign="top" align="left">2.42E&#x2212;02</td>
<td valign="top" align="left">Uncharacterized protein BB_0689</td>
</tr>
<tr>
<td valign="top" align="left">O51604</td>
<td valign="top" align="left">2.10E&#x2212;04</td>
<td valign="top" align="left">2.42E&#x2212;02</td>
<td valign="top" align="left">GTPase Era</td>
</tr>
<tr>
<td valign="top" align="left">O51141</td>
<td valign="top" align="left">2.61E&#x2212;04</td>
<td valign="top" align="left">2.53E&#x2212;02</td>
<td valign="top" align="left">Single-stranded DNA-binding protein</td>
</tr>
<tr>
<td valign="top" align="left">O51560</td>
<td valign="top" align="left">2.42E&#x2212;04</td>
<td valign="top" align="left">2.53E&#x2212;02</td>
<td valign="top" align="left">30S ribosomal protein S4</td>
</tr>
<tr>
<td valign="top" align="left">O51401</td>
<td valign="top" align="left">2.80E&#x2212;04</td>
<td valign="top" align="left">2.53E&#x2212;02</td>
<td valign="top" align="left">Fructose-bisphosphate aldolase</td>
</tr>
<tr>
<td valign="top" align="left">O51286</td>
<td valign="top" align="left">3.10E&#x2212;04</td>
<td valign="top" align="left">2.55E&#x2212;02</td>
<td valign="top" align="left">Ribosomal RNA small subunit methyltransferase H</td>
</tr>
<tr>
<td valign="top" align="left">O50821</td>
<td valign="top" align="left">3.23E&#x2212;04</td>
<td valign="top" align="left">2.55E&#x2212;02</td>
<td valign="top" align="left">Adenine deaminase</td>
</tr>
<tr>
<td valign="top" align="left">O51324</td>
<td valign="top" align="left">5.45E&#x2212;04</td>
<td valign="top" align="left">4.05E&#x2212;02</td>
<td valign="top" align="left">Uncharacterized protein</td>
</tr>
<tr>
<td valign="top" align="left">O50667</td>
<td valign="top" align="left">7.31E&#x2212;04</td>
<td valign="top" align="left">5.14E&#x2212;02</td>
<td valign="top" align="left">Type I restriction enzyme r protein n terminus (Hsdr_n)</td>
</tr>
<tr>
<td valign="top" align="left">P53362</td>
<td valign="top" align="left">8.00E&#x2212;04</td>
<td valign="top" align="left">5.32E&#x2212;02</td>
<td valign="top" align="left">tRNA uridine 5-carboxymethylaminomethyl modification enzyme MnmG</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>p-Values are not adjusted for multiple comparisons. FDR is p-value corrected for multiple comparisons using the Benjamini&#x2013;Hochberg method.</p>
</fn>
<fn>
<p>FDR, false discovery rate.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The second method for candidate protein selection utilizes the XGBoost classifier trained on the predicted binding intensities comparing the combined LD cohorts vs. the LAD cohort (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref>). The XGBoost algorithm is based on a decision tree structure and intrinsically performs feature selection during training. As a result, a trained classifier is based on a subset of the features (protein tiles, in this case) that contribute to the classification of the two cohorts. To identify protein candidates, first, only the tiles that were selected by the algorithm in all the cross-validation rounds (n = 10) were kept. Second, the <italic>B. burgdorferi</italic> proteins were then ranked by the number of different tiles they contained that met this criterion (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Data Sheet 2</bold></xref>). Proteins that contained at least three tiles from the list above were selected as candidate biomarkers, resulting in a total of 34 proteins (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>). The arbitrary threshold for the number of tiles per protein was set to 3 to both limit the number of candidate proteins and increase the likelihood of a protein showing true differentiating power.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Candidate biomarker proteins selected utilizing the classifier-based method.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">UniProt ID</th>
<th valign="top" align="left">Tile count</th>
<th valign="top" align="left">Protein</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">O51465</td>
<td valign="top" align="left">5</td>
<td valign="top" align="left">Uncharacterized protein</td>
</tr>
<tr>
<td valign="top" align="left">O51735</td>
<td valign="top" align="left">5</td>
<td valign="top" align="left">Outer membrane protein</td>
</tr>
<tr>
<td valign="top" align="left">O51067</td>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Uncharacterized protein BB_0038</td>
</tr>
<tr>
<td valign="top" align="left">O51157</td>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Transcription elongation factor GreA</td>
</tr>
<tr>
<td valign="top" align="left">O51291</td>
<td valign="top" align="left">4</td>
<td valign="top" align="left">NAD kinase</td>
</tr>
<tr>
<td valign="top" align="left">O51578</td>
<td valign="top" align="left">4</td>
<td valign="top" align="left">RecBCD enzyme subunit RecB</td>
</tr>
<tr>
<td valign="top" align="left">P42555</td>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Chaperone protein HtpG</td>
</tr>
<tr>
<td valign="top" align="left">O51319</td>
<td valign="top" align="left">4</td>
<td valign="top" align="left">DNA helicase</td>
</tr>
<tr>
<td valign="top" align="left">O51409</td>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Transporter, small conductance mechanosensitive ion channel (MscS) family</td>
</tr>
<tr>
<td valign="top" align="left">O51504</td>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Uncharacterized protein</td>
</tr>
<tr>
<td valign="top" align="left">O51195</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Uncharacterized protein BB_0173</td>
</tr>
<tr>
<td valign="top" align="left">O51229</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">DNA mismatch repair protein MutL</td>
</tr>
<tr>
<td valign="top" align="left">O51316</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Aspartyl/glutamyl-tRNA(Asn/Gln) amidotransferase subunit B</td>
</tr>
<tr>
<td valign="top" align="left">O51349</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">DNA-directed RNA polymerase subunit beta</td>
</tr>
<tr>
<td valign="top" align="left">O51540</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Arginine&#x2013;tRNA ligase</td>
</tr>
<tr>
<td valign="top" align="left">O51560</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">30S ribosomal protein S4</td>
</tr>
<tr>
<td valign="top" align="left">O51680</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Valine&#x2013;tRNA ligase</td>
</tr>
<tr>
<td valign="top" align="left">P50062</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Elongation factor Tu</td>
</tr>
<tr>
<td valign="top" align="left">P70838</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Uncharacterized protein BBD11</td>
</tr>
<tr>
<td valign="top" align="left">Q44737</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Chemotaxis protein CheA</td>
</tr>
<tr>
<td valign="top" align="left">G5IXI8</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Uncharacterized protein</td>
</tr>
<tr>
<td valign="top" align="left">H7C7M1</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">ErpM protein</td>
</tr>
<tr>
<td valign="top" align="left">O51310</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Oligopeptide transport system permease protein OppC</td>
</tr>
<tr>
<td valign="top" align="left">O51326</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Uncharacterized protein</td>
</tr>
<tr>
<td valign="top" align="left">O51381</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Sensory transduction histidine kinase, putative</td>
</tr>
<tr>
<td valign="top" align="left">O51485</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Uncharacterized protein</td>
</tr>
<tr>
<td valign="top" align="left">O51570</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">
<italic>N</italic>-acetylmuramoyl-L-alanine amidase, putative</td>
</tr>
<tr>
<td valign="top" align="left">O51574</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Pts system, fructose-specific IIABC component</td>
</tr>
<tr>
<td valign="top" align="left">O51655</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">zf-RING_7 domain-containing protein</td>
</tr>
<tr>
<td valign="top" align="left">O51687</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Oligopeptide ABC transporter, permease protein</td>
</tr>
<tr>
<td valign="top" align="left">O51770</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Exonuclease SbcC</td>
</tr>
<tr>
<td valign="top" align="left">O51774</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">ATP-dependent Clp protease, subunit C</td>
</tr>
<tr>
<td valign="top" align="left">O51784</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Lipoprotein, putative</td>
</tr>
<tr>
<td valign="top" align="left">Q9RZW8</td>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Adenine-specific DNA methyltransferase</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Tile count indicates the number of tiles from the corresponding protein that were selected by the classifier in each cross-validation round (n = 10).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Interestingly, neither of the lists contains any of the serologic biomarkers used in the current LD testing standard. This indicates that the differential humoral immune response between the LD and LAD cohorts may have a different set of target antigens than when comparing LD with healthy controls (<xref ref-type="bibr" rid="B36">36</xref>). The two lists show no overlap and represent two unique sets of proteins. The list produced by method 2 does contain several proteins [including the two top-ranked proteins in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>: the outer membrane protein (O51735) and an uncharacterized protein (O51465)] that are either known to be located on the membrane of the bacterium or are predicted extracellular (secreted) proteins. The cellular location makes these proteins accessible to antibody binding and provides further support for biological inference of the protein selection method. The finding that neither of the two lists contains any of the known LD biomarkers, e.g., the VlsE protein, is likely due to the difference in cohorts being compared. The standard serologic biomarker panel was developed to differentiate between LD and healthy controls, whereas here, LD was compared with LADs. Furthermore, the finding that the LD+ and LD&#x2212; groups of patients were characterized by distinct antibody reactivity profiles, as demonstrated previously (<xref ref-type="bibr" rid="B36">36</xref>) and in this study (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>), means that combining the two cohorts as conducted has the potential to result in a set of biomarkers that better differentiates the combined LD cohort from LADs.</p>
</sec>
<sec id="s2_5">
<title>Candidate biomarker verification using bead-based multiplex binding assays</title>
<p>A total of 53 candidate proteins from the <italic>B. burgdorferi</italic> proteome were selected for further consideration. However, only 44 of these were successfully synthesized; the nine remaining proteins were excluded from synthesis due to substantial transmembrane regions. In addition, the VlsE protein, a biomarker that is currently being used for standard serology testing for LD, was included as a positive control for the LD+ samples. For validation assays, the proteins were&#xa0;attached to carboxylated paramagnetic beads using <italic>N</italic>-hydroxysulfosuccinimide sodium salt (NHS) chemistry (Materials and Methods). The beads were incubated with serum samples, and the antibody binding was measured as fluorescence intensity using a secondary polyclonal anti-IgG antibody labeled with phycoerythrin. A total of 185 LD+, 102 LD&#x2212;, and 236 LAD samples were used for validation with the bead-based assays (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table&#xa0;1</bold>
</xref>). All samples were assayed in duplicate, and the mean values of the duplicates were used for further analysis. For data analysis, the binding values were converted to a log10 scale (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;3A</bold>
</xref>). The background binding signal was determined as the fluorescence intensity of the protein with the lowest coefficient of variation (CV) across all three cohorts. Low variation in the binding intensity of such a protein indicates that its binding is not disease-specific and can be used as a reference. While &#x201c;blank&#x201d; beads, i.e., beads not conjugated to any of the proteins, were also included in the assay as negative controls, it was observed that they showed some differential binding between the cohorts. It is possible that some of the antibodies in the three groups of patients exhibit preferential binding to the carboxyl moiety on the negative control beads. Further analyses, including classifier training, were performed with log-transformed intensity values that were used directly without further normalization. For classifier training, the LD+ and LD&#x2212; cohorts were separated and contrasted against the LAD cohort individually. This was conducted based on the earlier findings in this study that suggested a different humoral immune response profile in the two groups of patients (<xref ref-type="fig" rid="f1">
<bold>Figures&#xa0;1</bold>
</xref>-<xref ref-type="fig" rid="f4">
<bold>4</bold>
</xref>). One of the goals of the biomarker validation study was to determine if it would be possible to reduce the number of candidate proteins to a smaller subset to reduce the complexity of a potential diagnostic assay. To this end, classifier training was performed using all biomarkers and selected subsets. Feature selection was performed using two different methods: a) first, an XGBoost classifier was trained on data from all proteins. Because the algorithm performs feature selection during training, the proteins with the highest differentiation power (see <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;3B</bold>
</xref> for a cohort-level comparison of two highest ranking by importance proteins) were used for training another XGBoost classifier on the reduced set of features. b) Proteins were selected based on the p-values of a t-test whereby a number of proteins ranked by increasing p-value (decreasing statistical significance) were chosen for classifier training. <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5A&#x2013;C</bold>
</xref> show the receiver operating characteristic (ROC) curves of classifiers trained using the XGBoost algorithm and a subset of biomarkers that resulted in the best classification accuracy (highest AUC value) in the entire range of the number of proteins selected for training. The classifiers were trained to distinguish between three different contrasts: a combined LD+/LD&#x2212; cohort and LAD, LD+ vs. LAD, and LD&#x2212; vs. LAD patients. Classification accuracy in terms of AUC value was determined as a function of the number of proteins included in the training dataset. A comparison of the best classification performance achieved for each contrast when varying number of proteins in the panel was slightly lower for the combined LD vs. LADs with AUC = 0.84 (95 CI: 0.81&#x2013;0.88) (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5A</bold>
</xref>) than the two other contrasts with AUC values of 0.88 (95 CI: 0.84&#x2013;0.94) (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5B</bold>
</xref>) and 0.87 (95 CI: 0.82&#x2013;0.93) (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5C</bold>
</xref>) for LD+ vs. LAD and LD&#x2212; vs. LAD contrast, respectively. Classification performance as a function of the number of features (proteins) used in training (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5C&#x2013;E</bold>
</xref>) varies only slightly for the combined LD vs. LADs and LD+ vs. LADs. In comparison, the AUC values for the LD&#x2212; vs. LAD differentiation showed a slight but notable upward trend with the increasing number of features (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5D</bold>
</xref>). This suggests that the differentiating information is distributed more broadly in the LD&#x2212;/LAD than in the LD+/LAD contrast. A comparison of the AUC values obtained with the full set of proteins (n = 45) indicates that all three contrasts show a reduction in classification accuracy when the full set of proteins is used. Interestingly, the LD+ vs. LAD contrast shows improved performance with AUC = 0.88 (95 CI: 0.84&#x2013;0.94) when using 11 proteins that rank highest by importance/differentiating power as determined by the classifier trained on the full set of proteins (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5A</bold>
</xref>). While the LD&#x2212; vs. LAD classification gave a comparable performance to the LD+ vs. LAD classification when all features were used, there is one substantial difference between them. Interestingly, when comparing the biomarkers ranked by predictive power to distinguish between LD+/LADs and LD&#x2212;/LADs, it is clear that panel composition differs substantially between the two comparisons (<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>). Here, the gain parameter represents the fraction of overall classification performance that a particular protein biomarker contributes to the total classifier models shown in <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5B, C</bold>
</xref>. The higher values represent stronger differentiating power with the gain values of all biomarkers used adding up to 1. As expected, VlsE (UniProt ID G5IXI6) exhibits by far the most differentiating power (0.37) in separating LD+ from LAD patients. Consequently, the removal of VlsE from the training dataset resulted in a marked drop in the AUC value to 0.71 (data not shown). However, it is also clear that VlsE alone was not sufficient to achieve the demonstrated classification performance (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5B</bold>
</xref>) and that additional biomarkers made smaller but substantial contributions to the classification performance. In contrast, VlsE had a distinguishing power of only 0.05 in the LD&#x2212;/LAD contrast with three other proteins scoring higher. This is consistent with the LD&#x2212; cohort being negative on standard panels containing VlsE. Overall, the data in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> suggest differences in biomarker panels between the two contrasts, with VlsE being the only protein shared among the 10 highest-ranking biomarkers.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Classification performance of models trained on the bead-based assay data. Combined LD+/LD&#x2212; <bold>(A)</bold>, LD+ <bold>(B)</bold> and LD&#x2212; <bold>(C)</bold> cohorts were each compared against the sera from the patients in the LAD cohort using subsets of proteins that resulted in highest accuracy. The number of proteins used for training was varied for all three contrasts <bold>(D&#x2013;F)</bold> by selecting a subset of biomarker candidates either based on predictive power calculated by a classifier trained on a full set of proteins (blue curve) or by ranking the proteins based on the p-value (orange curve). LD, Lyme disease.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fimmu-16-1528524-g005.tif"/>
</fig>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Comparison of first 10 biomarkers ranked by predictive power (gain) between LD+/LAD and LD&#x2212;/LAD contrasts.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" rowspan="2" align="left">Protein number</th>
<th valign="top" colspan="2" align="left">LD+ vs. LADs</th>
<th valign="top" colspan="2" align="left">LD&#x2212; vs. LADs</th>
</tr>
<tr>
<th valign="top" align="left">UniProt ID</th>
<th valign="top" align="left">Gain</th>
<th valign="top" align="left">UniProt ID</th>
<th valign="top" align="left">Gain</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left">
<bold>
<italic>G5IXI6(VIsE)</italic>
</bold>
</td>
<td valign="top" align="left">
<bold>
<italic>0.370383</italic>
</bold>
</td>
<td valign="top" align="left">O51324</td>
<td valign="top" align="left">0.081201</td>
</tr>
<tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">O51784</td>
<td valign="top" align="left">0.045986</td>
<td valign="top" align="left">P50062</td>
<td valign="top" align="left">0.066731</td>
</tr>
<tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left">H7C7M1</td>
<td valign="top" align="left">0.029262</td>
<td valign="top" align="left">O51286</td>
<td valign="top" align="left">0.057566</td>
</tr>
<tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">P53362</td>
<td valign="top" align="left">0.029069</td>
<td valign="top" align="left">
<bold>
<italic>G5IXI6(VIsE)</italic>
</bold>
</td>
<td valign="top" align="left">
<bold>
<italic>0.05673</italic>
</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left">O51291</td>
<td valign="top" align="left">0.026474</td>
<td valign="top" align="left">O51555</td>
<td valign="top" align="left">0.054708</td>
</tr>
<tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left">O51353</td>
<td valign="top" align="left">0.026081</td>
<td valign="top" align="left">H7C7M1</td>
<td valign="top" align="left">0.04691</td>
</tr>
<tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left">P52323</td>
<td valign="top" align="left">0.024759</td>
<td valign="top" align="left">O50667</td>
<td valign="top" align="left">0.045187</td>
</tr>
<tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left">O51632</td>
<td valign="top" align="left">0.021474</td>
<td valign="top" align="left">O51229</td>
<td valign="top" align="left">0.0423</td>
</tr>
<tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left">O51326</td>
<td valign="top" align="left">0.020214</td>
<td valign="top" align="left">P53362</td>
<td valign="top" align="left">0.041274</td>
</tr>
<tr>
<td valign="top" align="left">10</td>
<td valign="top" align="left">O51655</td>
<td valign="top" align="left">0.018787</td>
<td valign="top" align="left">O51570</td>
<td valign="top" align="left">0.031742</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>LD, Lyme disease; LAD, look-alike disease.</p>
<p>The values in bold represent the VlsE protein, the main biomarker currently used in the standard LD serology.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Both feature selection methods (t-test and XGBoost) resulted in comparable classification outcomes, with the t-test-based selection method resulting in lower performance compared to the classifier-based method (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5C&#x2013;E</bold>
</xref>). The classification performance as a function of the number of selected proteins demonstrates that one can substantially reduce the number of antigens in the panel without markedly affecting classification performance. For example, the data in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5D</bold>
</xref> show that one can achieve comparable performance for differentiating between the combined LD and LAD patients with as few as five proteins, whereas only slightly reduced performance can be reached with five and six proteins for LD+ vs. LADs (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5E</bold>
</xref>) and LD&#x2212; vs. LADs (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5F</bold>
</xref>), respectively.</p>
<p>In summary, the validation assay data analysis suggests that one can accurately differentiate between the two LD cohorts and the look-alike diseases using a subset of protein biomarkers predicted to have differentiating power using the peptide array data. The most important finding is the fact that the LD&#x2212; patients who tested previously negative in the standard serology test can be reliably differentiated from the LAD patients.</p>
</sec>
</sec>
<sec id="s3" sec-type="discussion">
<title>Discussion</title>
<p>This study was designed to address the two following questions. First, in a general sense, can a broad, agnostic profiling of the humoral immune response using short, linear peptide libraries with randomly generated sequences that equally but sparsely sample an entire combinatorial space of peptides with the same length be utilized to extract biologically relevant information concerning immunogenic targets that the humoral immune system is responding to? The second, more specific question is focused on whether LD can be reliably differentiated from other diseases with similar clinical manifestations. Given the agnostic nature of the method and its unbiased approach to profiling circulating antibody binding, answering these questions would enable the evaluation of the method&#x2019;s ability to identify novel diagnostic biomarkers. Such a method has the potential to be used for answering similar questions for a broad range of diseases.</p>
<p>The humoral immune response profiling in the three cohorts reveals a heterogeneous picture in terms of antibody binding patterns. A comparison of the binding intensity distributions shows that, overall, the lowest antibody reactivity is in the LD&#x2212; group of patients as judged by the location of the second peak and the width of the distributions (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>). The substantially stronger response observed in the LD+ cohort may be due to a later timepoint in the acute LD stage when the samples were collected from these patients. The longer duration of immune system exposure to the pathogen in these patients may allow for a more robust adaptive immune response to be mounted against the bacterium, resulting in a stronger and more focused antibody response. It is also possible that the antibody response in LD&#x2212; patients is a result of the immunosuppressive mechanisms intrinsic to <italic>B. burgdorferi</italic> that subdue or abrogate an early response and result in more time for the bacterium to establish an infection. In comparison, the LAD binding distribution characteristics suggest a substantially stronger response, despite the fact that this cohort encompasses a number of different diseases. This difference is consistent with the notion of the immunosuppressive role of the bacterium when interacting with the host immune system. Despite the observed differences in the binding intensity distributions, the array peptide binding patterns of the three cohorts are similar, as evidenced by the UMAP representation of the binding data (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>). This finding underscores the difficulty in distinguishing LD from other diseases due to large patient-to-patient variability in humoral immune response and possibly partial cross-reactivity between antibodies raised against the different pathogens. Cross-reactive antibodies have been identified to overlap with LD-specific responses in Epstein&#x2013;Barr virus, <italic>Treponema pallidum</italic> infections (<xref ref-type="bibr" rid="B14">14</xref>&#x2013;<xref ref-type="bibr" rid="B16">16</xref>), and rheumatoid arthritis (<xref ref-type="bibr" rid="B46">46</xref>&#x2013;<xref ref-type="bibr" rid="B48">48</xref>). Previous publications from this lab and others have reported high levels of patient-to-patient variability in LD in terms of antibody reactivities (<xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B44">44</xref>). The results presented above are based on serum samples collected from several biobanks potentially increasing the patient-to-patient variation further due to differences in sample collection and storage protocols, testing, and different geographic locations. The use of ML models trained on the peptide array binding data enabled the learned sequence-binding relationship to be &#x201c;transferred&#x201d; onto a biologically relevant level by predicting antibody binding to a tiled <italic>B. burgdorferi</italic> proteome. The predicted binding intensities exhibited similar distribution characteristics as the peptide array intensities (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>), suggesting that binding pattern information as measured on the peptide array is properly captured on the proteome. Despite the somewhat reduced distinction among the hypothetical clusters of individuals observed in the binding data measured on the peptide arrays, the overall intensity distribution characteristics in terms of shape (<xref ref-type="fig" rid="f1">
<bold>Figures&#xa0;1A</bold>
</xref>, <xref ref-type="fig" rid="f3">
<bold>3A</bold>
</xref>) and UMAP representation (<xref ref-type="fig" rid="f1">
<bold>Figures&#xa0;1B</bold>
</xref>, <xref ref-type="fig" rid="f3">
<bold>3B</bold>
</xref>) between the measured and predicted data show close similarity. The binding predictions generated by NN models trained on each individual&#x2019;s data showed high accuracy (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure&#xa0;2</bold>
</xref>), supporting the validity of the chosen predictive ML models. These data further suggest that compared to the LD+ cohort, the LD&#x2212; cohort shows a distinct, although weaker, humoral response with possibly more pronounced person-to-person variability to the pathogen. The decreased overall binding is consistent with either being in the early stages of the infection or the inability to mount a response because of the immunosuppressive mechanisms engaged by the bacterium.</p>
<p>The observed similar classification performance between the measured and predicted binding values of models combining the LD+ and LD&#x2212; cohorts (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2D</bold>
</xref>, <xref ref-type="fig" rid="f4">
<bold>4D</bold>
</xref>) as opposed to the classifiers contrasting the two cohorts against the LAD patients separately (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2E, F</bold>
</xref>, <xref ref-type="fig" rid="f4">
<bold>4E, F</bold>
</xref>, respectively) indicates that the differentiating power of the two classifiers is comparable. This result can likely be explained by the increased sample size in the combined LD cohort that leads to better classification model generalization and consistency that are less dependent on the source of the training data (measured vs. predicted). The finding further suggests the importance of having adequately sized patient cohorts in LD studies where patient-to-patient variability is notable and needs to be taken into account.</p>
<p>Importantly, the difference in the biomarker panels in terms of contributions to classification performance (<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>) implies that the pathogen-specific antibody reactivity profile in LD&#x2212; patients is different from that in the LD+ group. The only overlap between the two panels is the VlsE protein, yet its relative contributions to the overall differentiating power is approximately an order of magnitude less in the LD&#x2212; cohort analysis. There is no other overlap among the nine remaining biomarkers. Also, the predictive power in the LD&#x2212;/LAD contrast appears distributed more evenly among the panel biomarkers, suggesting a more dispersed humoral response in the LD&#x2212; cohort compared to the LD+ cohort. The difference in the antibody reactivity profiles is also consistent with the negative standard serologic testing outcome for the LD&#x2212; patients and may explain why no antibody reactivity to the biomarkers used in the standard serologic LD test is present. Nevertheless, despite the lack of antibody response to the current standard of testing, the findings of this study demonstrate that in these patients, there is an ongoing, <italic>B. burgdorferi</italic>-specific humoral immune response toward a set of immunogenic targets that are different from those typically found in seropositive LD patients using the current testing standard.</p>
<p>Interestingly, it was found that the classifiers trained on the predicted binding values of <italic>B. burgdorferi</italic> protein tiled sequences (linear 10 AA long peptides with 9 AA overlap) to distinguish between LD+ or LD&#x2212; and LADs (<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4E, D</bold>
</xref>) showed substantially lower differentiating accuracy than the classifiers trained on binding data obtained with full recombinant proteins in the bead-based assays (<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5B, C</bold>
</xref>). Note that the proteins for the assays were selected based on the binding to the tiled proteins from the <italic>B. burgdorferi</italic> proteome. This suggests that the &#x201c;hit&#x201d; tiles identified in the analysis truly belong to proteins that are targeted by the humoral immune response. These tiles serve as linear proxies of binding to fully assembled proteins that, once expressed, show substantially stronger differential binding than the linear tiles due to the presence of full structural epitopes. One possible explanation is that a fraction of the linear protein tiles &#x201c;hits&#x201d; may contain short (3&#x2013;6 amino acid long) motifs or their mimotopes (sequences that differ in amino acid arrangement but show similar physicochemical properties to the actual motif) that represent different linear portions of the same structural epitope(s) of a protein. If true, antibody binding to the fully assembled structural epitope on the protein would be substantially stronger and show a larger differential signal than the separate linear motifs of the peptides. The substantial increase in classification accuracy with full proteins therefore implies that at least some of the protein structure-specific binding is captured with the linear peptide arrays.</p>
<p>The fact that the peptide libraries are based on short peptides without any particular structural information is a major limitation given that the majority of peptide epitopes are structural and discontinuous. Nevertheless, earlier work from this group has demonstrated the utility of the approach to distinguish with high accuracy between a number of different diseases based simply on the binding patterns of antibodies contained in the blood (<xref ref-type="bibr" rid="B43">43</xref>). This study provides further support for the notion that some of the structural epitope information may be contained in the linear array binding data through the representation of the linear fragments that make up some of the structural epitopes. As a result, one may be able to utilize the binding to linear peptide data to identify immunogens containing either linear or a combination of linear and structural epitopes targeted by the immune system in response to pathogen infection.</p>
<p>In conclusion, the findings of this study highlight several challenges one is faced with when distinguishing LD from other diseases with similar clinical manifestations, especially in the early stages of LD. Strong patient-to-patient variability in the humoral immune response to <italic>B. burgdorferi</italic> combined with previously demonstrated cross-reactivity of antibodies raised in response to other pathogens both act as confounding factors in distinguishing LD from other LADs. Nevertheless, it was possible to identify and validate a panel of biomarkers that robustly differentiates between seropositive or seronegative LD and LADs. Furthermore, the results suggest a different humoral immune response profile in the LD+ and LD&#x2212; patients through the finding of separate panels of biomarkers that are specific to each condition. Perhaps most importantly, the ability to distinguish between seronegative LD patients and LADs is especially valuable, as these patients would have been deemed non-LD and either misdiagnosed with another disease or subjected to further unnecessary testing. The study findings also corroborate the notion that combining data of polyclonal antibody binding to a library of linear peptides with machine learning models provides biologically relevant information about the humoral immune response underlying an acute infection with the <italic>B. burgdorferi</italic> bacterium. Due to the agnostic nature of the approach, it is also likely that the method can be utilized for profiling the humoral response and biomarker discovery for a number of other diseases with a strong humoral immune system involvement.</p>
</sec>
<sec id="s4" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="s4_1">
<title>Samples</title>
<p>Seropositive LD (LD+) patient serum samples were obtained from the Lyme Disease Biobank Foundation, Portland, OR (<xref ref-type="bibr" rid="B3">3</xref>), CDC, and several commercial biobanks (Boca Biolistics, Pompano Beach, FL; Discovery Life Sciences, Huntsville, AL; and SeraCare, Milford, MA). Samples were collected from patients with signs and symptoms of LD. Samples were tested using the STTT and categorized as seropositive Lyme disease having either an EM rash greater than 5&#xa0;cm in diameter or PCR confirmation combined with positive STTT serology. The seronegative Lyme disease samples were obtained from patients having an EM rash greater than 5&#xa0;cm in diameter, but without positive STTT serology, and were obtained from the Lyme Disease Biobank Foundation. These patients were diagnosed with Lyme disease by a physician based on the patients&#x2019; clinical symptoms. Participants were enrolled in East Hampton, NY, Central Wisconsin, and Martha&#x2019;s Vineyard, MA. Each of the three cohorts contained equivalent numbers of patients from each collection site. Cohorts and collection sites were balanced across each assay batch of microarrays. The patient samples for the diseases with similar etiology to LD (look-alike diseases) were obtained from several commercial sources (Boca Biolistics, Pompano Beach, FL; Discovery Life Sciences, Huntsville, AL; Creative Testing Solutions, Tempe, AZ; and SeraCare, Milford, MA). Note that these samples were obtained from commercial biobanks, with no data about their previous exposure to LD provided. Given that the samples were collected outside of the LD endemic areas, it is unlikely that these patients have been exposed to <italic>B. burgdorferi</italic> infection.</p>
</sec>
<sec id="s4_2">
<title>Peptide microarray assays</title>
<p>Peptide microarrays containing diverse peptides were synthesized in a commercial production facility (Cowper Sciences, Chandler, AZ), following a previously described library design and photo-lithography-based manufacturing process (<xref ref-type="bibr" rid="B34">34</xref>, <xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B37">37</xref>). Briefly, the microarrays used contained 126,051 diverse peptides plus a set of 6,203 control peptides of varying lengths ranging from 6 to 13&#x2013; amino acids. The standard serum Ab profiling assay protocol described by Arvey et&#xa0;al. (<xref ref-type="bibr" rid="B34">34</xref>) was used and modified for a modular research use assay system as described by Kelbauskas et&#xa0;al. (<xref ref-type="bibr" rid="B36">36</xref>). Samples were thawed from single-use aliquots and diluted to 1:625 in assay buffer (Phosphate-buffered saline/Tween (PBST) with 0.05% Tween 20, 0.1% ProClin 950, and 1% mannitol). Diluted samples (90 &#x3bc;L) were applied to arrays and incubated for 1&#xa0;h at 37&#xb0;C with mixing (TeleShake 95 platform mixer). The cassette was then washed three times in PBST-P using a 96-well microtiter plate washer (BioTek Instruments, Inc., Winooski, VT). Peptide-bound serum antibodies were detected using either 4.0 nM goat anti-human IgG (H+L) conjugated to AlexaFluor 555 (Invitrogen&#x2013;Thermo Fisher Scientific, Inc., Carlsbad, CA) or 4.0 nM goat anti-human IgM (H+L) (Novus Biologicals, Centennial, CO), conjugated to DyLight 550 in secondary incubation buffer (0.5% casein in PBST-P) for 1&#xa0;h with mixing at 37&#xb0;C. After the final incubation, slides were washed three times with PBST-P followed by distilled water to remove residual salts. Slides were then sprayed with isopropanol and dried by centrifugation.</p>
</sec>
<sec id="s4_3">
<title>Peptide microarray data extraction</title>
<p>Dried slides were imaged using an ImageXpress imaging system to detect fluorescently labeled secondary antibodies. The imager used an LED light engine (SemRock) centered at 532-nm wavelength to excite fluorophore-conjugated secondary Ab. Mapix (version 7.2.1; Innopsys, Carbonne, France) was used to place a grid alignment file over the obtained images and extract the median foreground pixel intensities using the central 60% of each feature.</p>
</sec>
<sec id="s4_4">
<title>Data quality checks</title>
<p>Images were inspected to identify arrays with artifacts and image anomalies. The samples associated with such arrays were re-assayed on arrays from the same production batch as the original assay.</p>
</sec>
<sec id="s4_5">
<title>Modeling of peptide binding using machine learning</title>
<p>Predictive models were built using machine learning methods based on feed-forward, backpropagating fully connected neural networks, similar to those described previously (<xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B42">42</xref>). The peptide sequence was one-hot encoded by transforming each peptide into a vector of length 190. The vector length was derived from a maximum peptide length of 10 residue positions with 19 possible amino acids for each position. Feed-forward neural networks were built individually for each donor using R (version 4.2.2, R Foundation for Statistical Computing, Vienna, Austria) as the programming language and utilizing TensorFlow (version 2.11.0) and Keras (version 2.11.1) as the interface packages. The NN models were constructed using three hidden layers with 100 nodes each with a 10% dropout and no layer bias. Rectified linear unit (RelU) activation was used for each layer. Each NN model was trained 10 times using a random 90:10 split of the dataset each time. The data points were weighted by the frequency of peptides appearing in an intensity interval. To this end, the entire intensity range was subdivided into 100 equal bins, and the number of peptides falling into each bin was calculated. The weight for each peptide was computed using the following formula:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>w<sub>i</sub>
</italic> is the weight of the <italic>i</italic>th peptide and <italic>n<sub>i</sub>
</italic> is the number of peptides in the bin that the <italic>i</italic>th peptide falls into.</p>
<p>The accuracy of the model to predict Ab binding to the array was evaluated by predicting the binding to the held-out 10% of the data and reported as Pearson&#x2019;s correlation between the measured and predicted binding intensities. Binding to <italic>B. burgdorferi</italic> epitopes was accomplished by applying the NN models to the <italic>B. burgdorferi</italic> B31 reference proteome (UniProt Accession # UP000001807) that had been represented as 10-mers with a sliding window of one amino acid offsets.</p>
</sec>
<sec id="s4_6">
<title>Classification</title>
<p>Random forest decision trees with XGBoost were used to train classifiers for distinguishing patients from the different cohorts used. Each classifier model was trained 10 times on randomly selected 90% of the patients from the corresponding cohorts, and its performance was assessed on the remaining 10% of patients. All training steps and mean receiver operating characteristic curve calculations were performed in R.</p>
</sec>
<sec id="s4_7">
<title>Outlier sum statistics</title>
<p>The outlier sum statistics was implemented following the method published by Tibshirani et&#xa0;al. (<xref ref-type="bibr" rid="B45">45</xref>). The predicted binding intensity values were first z-score normalized using the following:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>I<sub>i</sub>
</italic>,<italic>
<sub>j</sub>
</italic> is the predicted binding intensity value of the <italic>j</italic>th tile in the <italic>i</italic>th sample, and mean and SD are the mean and standard deviation values of <italic>I<sub>i</sub>
</italic>, respectively. In this way, binding intensity values were all normalized to their corresponding mean and standard deviation values. Next, the Z-scores of the binding intensity values of the LD+ and LD&#x2212; samples (&#x201c;cases&#x201d;) were calculated using the means and SD values of the LAD samples (&#x201c;controls&#x201d;):</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>I<sub>j</sub>
</italic>
<sub>(</sub>
<italic>
<sub>c</sub>
</italic>
<sub>)</sub> denotes the predicted binding intensities of the <italic>j</italic>th tile in the control samples. Afterward, a sliding window smoothing with a window size of 5 was applied to the data. Next, for each protein from the <italic>B. burgdorferi</italic> proteome, the tile with the maximum <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> value for each patient in the case cohort was determined. As a result, each protein is represented as its maximum binding value normalized against the control samples as a reference. These maximum binding values were then used to compute the outlier sum (OS) statistics for each protein:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mo>&gt;</mml:mo>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mrow>
<mml:mn>0.75</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>p</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>Q</mml:mi>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>p</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>N</italic> is the number of samples in the case cohort, <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents the maximum binding value for protein <italic>p</italic> in sample <italic>i</italic>, <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>p</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the maximum binding value of protein <italic>p</italic> of the samples in the case cohort, and <italic>q</italic>
<sub>0.75</sub> and IQR are the third quartile and interquartile range of <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msubsup>
<mml:mi>I</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, respectively. The p-values for each <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> were calculated using the t-test, and a null distribution was obtained by randomizing the cohort assignments of the samples 1,000 times. The false discovery rate was calculated using the Benjamini&#x2013;Hochberg adjustment for multiple comparisons.</p>
</sec>
<sec id="s4_8">
<title>Bead-based assays and data analysis</title>
<p>The functionalization of Luminex MagPlex (Diasorin, Madison, WI) microspheres (beads) was performed by reacting the carboxylic residues of the microspheres and amine groups of proteins using 1-(3-(dimethylamino)propyl)-3-ethyl-carbodiimide hydrochloride (EDAC)/NHS chemistry. Briefly, 236 &#x3bc;L of each address of stock Luminex MagPlex microspheres (1.27 &#xd7; 10<sup>7</sup> beads/mL) was resuspended into 764 &#x3bc;L deionized water (DW; 18.2 M&#x3a9;&#xb7;cm), followed by washing with 1 mL of DW. A total of 46 different regions of microspheres each representing a different spectral region for multiplex detection were used for the binding assays. The carboxylic residues of the microspheres were activated by incubating with 90 &#x3bc;L of 50 mg/mL of NHS (Sigma-Aldrich, St. Louis, MO) and 90 &#x3bc;L of 50 mg/mL of EDAC (Sigma-Aldrich) in 1 mL of 0.1 M sodium phosphate (Sigma-Aldrich) buffer for 20&#xa0;min at room temperature (RT) under gentle rotation. For each bead region, the reaction using 106 microspheres and 5 &#x3bc;g of protein was carried out in 900 &#x3bc;L of 50 mM of MES (pH 5.0) buffer (Avocado Research Chemicals Ltd., Heysham, Lancashire, UK) for 2&#xa0;h with gentle rotation at RT. After removing the supernatant, the functionalized microspheres were resuspended into 1 mL of PBS-TBN [PBS buffer (Life Technologies, Burlington, ON, Canada) with 0.02% Tween-20 (Sigma-Aldrich), 0.1% bovine serum albumin (BSA; Sigma-Aldrich), 0.02% sodium azide (Sigma-Aldrich), 150 mM sodium chloride (Life Technologies), and 50 mM sodium phosphate monobasic, pH 7.4]. The surface was then blocked with 1% BSA by rotating for 30&#xa0;min at RT. The beads were washed three times with PBST. Next, all 46 regions of the functionalized microspheres were combined and mixed with 54 mL of PBST-BSA [PBS buffer with 0.1% Tween-20 (Sigma-Aldrich) and 1% BSA] buffer for further use. Microspheres functionalized with the VlsE protein served as positive control for the LD+ cohort, and &#x201c;blank&#x201d; beads that went through the same preparation steps, but were not functionalized with a protein, served as negative control. All washing and supernatant removal steps were performed using a MagJET separation rack (Thermo Fisher Scientific, Carlsbad, CA) to separate the microspheres from the solution.</p>
<p>For assays, 50 &#x3bc;L of functionalized microspheres at a concentration of 40 beads/&#x3bc;L (a total of 2,000 beads) was first dispensed into each well of 96-well a non-binding 96-well plate with a flat bottom (Corning, Corning, NY) using a Bravo automated liquid dispensing system (Agilent, Santa Clara, CA). This step was followed by incubation with 50 &#x3bc;L of serum sample (diluted at 1:500 in PBST) for 1&#xa0;h at 37&#xb0;C with shaking at 500 rpm. After washing three times with a magnetic microplate washer (Biotek 405 TS, Agilent), 100 &#x3bc;L of goat anti-human IgG secondary antibody (Jackson ImmunoResearch, West Grove, PA) diluted to 1:125 was added to each well and incubated for 30&#xa0;min at RT with shaking at 500 rpm. After incubation, the beads were washed three times and resuspended in 100 &#x3bc;L PBST buffer. Binding signal intensities were then measured using a Luminex&#x2122; xMAP&#x2122; IntelliFlex (Diasorin) system.</p>
<p>The binding intensity values for each sample measured in the bead-based assays were normalized by computing the ratio between the intensity values and the intensity of a reference protein. The reference protein (O51141) was selected as a protein having the lowest CV when measured across all cohorts. The assumption was made that such a protein is least affected by the cohort-specific differences in antibody repertoires and thus can be used as a reference to compare binding intensities across patients.</p>
</sec>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are publicly available. This data can be found here: <uri xlink:href="https://figshare.com/articles/dataset/SI_table4_raw_array_data_csv/28299398">https://figshare.com/articles/dataset/SI_table4_raw_array_data_csv/28299398</uri>.</p>
</sec>
<sec id="s6" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Institutional Review Board at Arizona State University. The studies were conducted in accordance with the local legislation and institutional requirements. The human samples used in this study were acquired from a by-product of routine care or industry. Written informed consent for participation was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>LK: Conceptualization, Data curation, Formal Analysis, Funding acquisition, Investigation, Methodology, Project administration, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. TZ: Data curation, Formal Analysis, Validation, Visualization, Writing &#x2013; original draft. LB: Resources, Validation, Writing &#x2013; review &amp; editing. NW: Conceptualization, Methodology, Project administration, Resources, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This study was supported by a grant from the Department of Defence, Congressionally Directed Medical Research Program (CDMRP), grant #W81XWH-22-1-0204 (NW and LK).</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>All authors confirm that the following manuscript is a transparent and honest account of the reported research. This research is related to a previous study by the same authors titled Highly heterogenous humoral immune response in Lyme disease patients revealed by broad machine learning-assisted antibody binding profiling with random peptide arrays. The previous study was performed with the aim of assessing individual-to-individual variability in antibody reactivity in response to infection and validating biomarkers that distinguish acute Lyme disease from endemic healthy controls while the current submission is focusing on the discovery and validation of protein biomarkers for differentiation between Lyme disease and other febrile illnesses with clinical manifestations similar to those of Lyme disease. The study is in part following the methodology explained in the publication referenced above.</p>
</ack>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Author LK is the owner of Biomorph Technologies and may benefit from the publication of results presented here.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fimmu.2025.1528524/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fimmu.2025.1528524/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.xlsx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"/>
<supplementary-material xlink:href="Table2.xlsx" id="SM2" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"/>
<supplementary-material xlink:href="Table3.xlsx" id="SM3" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"/>
<supplementary-material xlink:href="Table4.xlsx" id="SM4" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Davidsson</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>The financial implications of a well-hidden and ignored chronic lyme disease pandemic</article-title>. <source>Healthcare (Basel)</source>. (<year>2018</year>) <volume>6</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/healthcare6010016</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="web">
<person-group person-group-type="author">
<collab>Centers for Disease Control and Prevention</collab>
</person-group>. Available online at: <uri xlink:href="https://www.cdc.gov/lyme/index.html">https://www.cdc.gov/lyme/index.html</uri> (Accessed <access-date>August 28, 2020</access-date>).</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Horn</surname> <given-names>EJ</given-names>
</name>
<name>
<surname>Dempsey</surname> <given-names>G</given-names>
</name>
<name>
<surname>Schotthoefer</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Prisco</surname> <given-names>UL</given-names>
</name>
<name>
<surname>McArdle</surname> <given-names>M</given-names>
</name>
<name>
<surname>Gervasi</surname> <given-names>&#xa0;
</given-names>
</name>
<etal/>
</person-group>. <article-title>The lyme disease biobank: characterization of 550 patient and control samples from the east coast and upper midwest of the United States</article-title>. <source>J Clin Microbiol</source>. (<year>2020</year>) <volume>58</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/JCM.00032-20</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mazori</surname> <given-names>DR</given-names>
</name>
<name>
<surname>Orme</surname> <given-names>CM</given-names>
</name>
<name>
<surname>Mir</surname> <given-names>A</given-names>
</name>
<name>
<surname>Meehan</surname> <given-names>SA</given-names>
</name>
<name>
<surname>Neimann</surname> <given-names>AL</given-names>
</name>
</person-group>. <article-title>Vesicular erythema migrans: an atypical and easily misdiagnosed form of Lyme disease</article-title>. <source>Dermatol Online J</source>. (<year>2015</year>) <volume>21</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.5070/D3218028428</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paul</surname> <given-names>S</given-names>
</name>
<name>
<surname>Song</surname> <given-names>PI</given-names>
</name>
<name>
<surname>Ogbechie</surname> <given-names>OA</given-names>
</name>
<name>
<surname>Sugai</surname> <given-names>DY</given-names>
</name>
<name>
<surname>Morley</surname> <given-names>KW</given-names>
</name>
<name>
<surname>Schalock</surname> <given-names>PC</given-names>
</name>
<etal/>
</person-group>. <article-title>Vesiculobullous and hemorrhagic erythema migrans: uncommon variants of a common disease</article-title>. <source>Int J Dermatol</source>. (<year>2016</year>) <volume>55</volume>:<page-range>e79&#x2013;82</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/ijd.2016.55.issue-2</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Steere</surname> <given-names>AC</given-names>
</name>
<name>
<surname>Sikand</surname> <given-names>VK</given-names>
</name>
</person-group>. <article-title>The presenting manifestations of Lyme disease and the outcomes of treatment</article-title>. <source>N Engl J Med</source>. (<year>2003</year>) <volume>348</volume>:<page-range>2472&#x2013;4</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1056/NEJM200306123482423</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Masters</surname> <given-names>E</given-names>
</name>
<name>
<surname>Granter</surname> <given-names>S</given-names>
</name>
<name>
<surname>Duray</surname> <given-names>P</given-names>
</name>
<name>
<surname>Cordes</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>Physician-diagnosed erythema migrans and erythema migrans-like rashes following Lone Star tick bites</article-title>. <source>Arch Dermatol</source>. (<year>1998</year>) <volume>134</volume>:<page-range>955&#x2013;60</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/archderm.134.8.955</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wormser</surname> <given-names>GP</given-names>
</name>
<name>
<surname>Masters</surname> <given-names>E</given-names>
</name>
<name>
<surname>Nowakowski</surname> <given-names>J</given-names>
</name>
<name>
<surname>McKenna</surname> <given-names>D</given-names>
</name>
<name>
<surname>Holmgren</surname> <given-names>D</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Prospective clinical evaluation of patients from Missouri and New York with erythema migrans-like skin lesions</article-title>. <source>Clin Infect Dis</source>. (<year>2005</year>) <volume>41</volume>:<page-range>958&#x2013;65</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1086/432935</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Seidel</surname> <given-names>MF</given-names>
</name>
<name>
<surname>Domene</surname> <given-names>AB</given-names>
</name>
<name>
<surname>Vetter</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>Differential diagnoses of suspected Lyme borreliosis or post-Lyme-disease syndrome</article-title>. <source>Eur J Clin Microbiol Infect Dis</source>. (<year>2007</year>) <volume>26</volume>:<page-range>611&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10096-007-0342-0</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hirsch</surname> <given-names>AG</given-names>
</name>
<name>
<surname>Herman</surname> <given-names>RJ</given-names>
</name>
<name>
<surname>Rebman</surname> <given-names>A</given-names>
</name>
<name>
<surname>Moon</surname> <given-names>KA</given-names>
</name>
<name>
<surname>Aucott</surname> <given-names>J</given-names>
</name>
<name>
<surname>Heaney</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Obstacles to diagnosis and treatment of Lyme disease in the USA: a qualitative study</article-title>. <source>BMJ Open</source>. (<year>2018</year>) <volume>8</volume>:<elocation-id>e021367</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/bmjopen-2017-021367</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Belongia</surname> <given-names>EA</given-names>
</name>
<name>
<surname>Reed</surname> <given-names>KD</given-names>
</name>
<name>
<surname>Mitchell</surname> <given-names>PD</given-names>
</name>
<name>
<surname>Mueller-Rizner</surname> <given-names>N</given-names>
</name>
<name>
<surname>Vandermause</surname> <given-names>M</given-names>
</name>
<name>
<surname>Finkel</surname> <given-names>MF</given-names>
</name>
<etal/>
</person-group>. <article-title>Tickborne infections as a cause of nonspecific febrile illness in Wisconsin</article-title>. <source>Clin Infect Dis</source>. (<year>2001</year>) <volume>32</volume>:<page-range>1434&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1086/320160</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aucott</surname> <given-names>JN</given-names>
</name>
<name>
<surname>Seifter</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Misdiagnosis of early Lyme disease as the summer flu</article-title>. <source>Orthop Rev (Pavia)</source>. (<year>2011</year>) <volume>3</volume>:<elocation-id>e14</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.4081/or.2011.e14</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koester</surname> <given-names>TM</given-names>
</name>
<name>
<surname>Meece</surname> <given-names>JK</given-names>
</name>
<name>
<surname>Fritsche</surname> <given-names>TR</given-names>
</name>
<name>
<surname>Frost</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>Infectious mononucleosis and lyme disease as confounding diagnoses: A report of 2 cases</article-title>. <source>Clin Med Res</source>. (<year>2018</year>) <volume>16</volume>:<page-range>66&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3121/cmr.2018.1419</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wojciechowska-Koszko</surname> <given-names>I</given-names>
</name>
<name>
<surname>Kwiatkowski</surname> <given-names>P</given-names>
</name>
<name>
<surname>Sienkiewicz</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kowalczyk</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kowalczyk</surname> <given-names>E</given-names>
</name>
<name>
<surname>Do&#x142;&#x119;gowska</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Cross-reactive results in serological tests for borreliosis in patients with active viral infections</article-title>. <source>Pathogens</source>. (<year>2022</year>) <volume>11</volume>:<fpage>203</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/pathogens11020203</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Panelius</surname> <given-names>J</given-names>
</name>
<name>
<surname>Lahdenne</surname> <given-names>P</given-names>
</name>
<name>
<surname>Heikkila</surname> <given-names>T</given-names>
</name>
<name>
<surname>Peltomaa</surname> <given-names>M</given-names>
</name>
<name>
<surname>Oksi</surname> <given-names>J</given-names>
</name>
<name>
<surname>Seppala</surname> <given-names>I</given-names>
</name>
</person-group>. <article-title>Recombinant OspC from Borrelia burgdorferi sensu stricto, B. afzelii and B. garinii in the serodiagnosis of Lyme borreliosis</article-title>. <source>J Med Microbiol</source>. (<year>2002</year>) <volume>51</volume>:<page-range>731&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1099/0022-1317-51-9-731</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Magnarelli</surname> <given-names>LA</given-names>
</name>
<name>
<surname>Lawrenz</surname> <given-names>M</given-names>
</name>
<name>
<surname>Norris</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Fikrig</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>Comparative reactivity of human sera to recombinant VlsE and other Borrelia burgdorferi antigens in class-specific enzyme-linked immunosorbent assays for Lyme borreliosis</article-title>. <source>J Med Microbiol</source>. (<year>2002</year>) <volume>51</volume>:<page-range>649&#x2013;55</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1099/0022-1317-51-8-649</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ryffel</surname> <given-names>K</given-names>
</name>
<name>
<surname>P&#xe9;ter</surname> <given-names>O</given-names>
</name>
<name>
<surname>Binet</surname> <given-names>L</given-names>
</name>
<name>
<surname>Dayer</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>Interpretation of immunoblots for Lyme borreliosis using a semiquantitative approach</article-title>. <source>Clin Microbiol Infect</source>. (<year>1998</year>) <volume>4</volume>:<page-range>205&#x2013;12</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1469-0691.1998.tb00670.x</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Radolf</surname> <given-names>J</given-names>
</name>
<name>
<surname>Samuels</surname> <given-names>D</given-names>
</name>
</person-group>. <source>Lyme Disease and Relapsing Fever Spirochetes: Genomics. Molecular Biology, Host Interactions and Disease Pathogenesis</source>. <publisher-loc>Norfolk, UK</publisher-loc>: <publisher-name>Caister Academic Press</publisher-name> (<year>2021</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.21775/9781913652616</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barbour</surname> <given-names>AG</given-names>
</name>
<name>
<surname>Travinsky</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Evolution and distribution of the ospC Gene, a transferable serotype determinant of Borrelia burgdorferi</article-title>. <source>mBio</source>. (<year>2010</year>) <volume>1</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/mBio.00153-10</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schutzer</surname> <given-names>SE</given-names>
</name>
<name>
<surname>Fraser-Liggett</surname> <given-names>CM</given-names>
</name>
<name>
<surname>Casjens</surname> <given-names>SR</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>WG</given-names>
</name>
<name>
<surname>Dunn</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Mongodin</surname> <given-names>EF</given-names>
</name>
<etal/>
</person-group>. <article-title>Whole-genome sequences of thirteen isolates of Borrelia burgdorferi</article-title>. <source>J Bacteriol</source>. (<year>2011</year>) <volume>193</volume>:<page-range>1018&#x2013;20</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/JB.01158-10</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Casjens</surname> <given-names>S</given-names>
</name>
<name>
<surname>Palmer</surname> <given-names>N</given-names>
</name>
<name>
<surname>van Vugt</surname> <given-names>R</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>WM</given-names>
</name>
<name>
<surname>Stevenson</surname> <given-names>B</given-names>
</name>
<name>
<surname>Rosa</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>A bacterial genome in flux: the twelve linear and nine circular extrachromosomal DNAs in an infectious isolate of the Lyme disease spirochete Borrelia burgdorferi</article-title>. <source>Mol Microbiol</source>. (<year>2000</year>) <volume>35</volume>:<fpage>490</fpage>&#x2013;<lpage>516</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1046/j.1365-2958.2000.01698.x</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Castellanos</surname> <given-names>M</given-names>
</name>
<name>
<surname>Verhey</surname> <given-names>TB</given-names>
</name>
<name>
<surname>Chaconas</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>A Borrelia burgdorferi mini-vls system that undergoes antigenic switching in mice: investigation of the role of plasmid topology and the long inverted repeat</article-title>. <source>Mol Microbiol</source>. (<year>2018</year>) <volume>109</volume>:<page-range>710&#x2013;21</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/mmi.2018.109.issue-5</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Verhey</surname> <given-names>TB</given-names>
</name>
<name>
<surname>Castellanos</surname> <given-names>M</given-names>
</name>
<name>
<surname>Chaconas</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>Analysis of recombinational switching at the antigenic variation locus of the Lyme spirochete using a novel PacBio sequencing pipeline</article-title>. <source>Mol Microbiol</source>. (<year>2018</year>) <volume>108</volume>:<fpage>461</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/mmi.2018.108.issue-4</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Verhey</surname> <given-names>TB</given-names>
</name>
<name>
<surname>Castellanos</surname> <given-names>M</given-names>
</name>
<name>
<surname>Chaconas</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>Antigenic variation in the Lyme spirochete: detailed functional assessment of recombinational switching at vlsE in the JD1 strain of Borrelia burgdorferi</article-title>. <source>Mol Microbiol</source>. (<year>2019</year>) <volume>111</volume>:<page-range>750&#x2013;63</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/mmi.2019.111.issue-3</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ogden</surname> <given-names>NH</given-names>
</name>
<name>
<surname>Arsenault</surname> <given-names>J</given-names>
</name>
<name>
<surname>Hatchette</surname> <given-names>TF</given-names>
</name>
<name>
<surname>Mechai</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lindsay</surname> <given-names>LR</given-names>
</name>
</person-group>. <article-title>Antibody responses to Borrelia burgdorferi detected by western blot vary geographically in Canada</article-title>. <source>PloS One</source>. (<year>2017</year>) <volume>12</volume>:<elocation-id>e0171731</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0171731</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wormser</surname> <given-names>GP</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>AT</given-names>
</name>
<name>
<surname>Schimmoeller</surname> <given-names>NR</given-names>
</name>
<name>
<surname>Bittker</surname> <given-names>S</given-names>
</name>
<name>
<surname>Cooper</surname> <given-names>D</given-names>
</name>
<name>
<surname>Visintainer</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Utility of serodiagnostics designed for use in the United States for detection of Lyme borreliosis acquired in Europe and vice versa</article-title>. <source>Med Microbiol Immunol</source>. (<year>2014</year>) <volume>203</volume>:<fpage>65</fpage>&#x2013;<lpage>71</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00430-013-0315-0</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garg</surname> <given-names>K</given-names>
</name>
<name>
<surname>Meril&#xe4;inen</surname> <given-names>L</given-names>
</name>
<name>
<surname>Franz</surname> <given-names>O</given-names>
</name>
<name>
<surname>Pirttinen</surname> <given-names>H</given-names>
</name>
<name>
<surname>Quevedo-Diaz</surname> <given-names>M</given-names>
</name>
<name>
<surname>Croucher</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Evaluating polymicrobial immune responses in patients suffering from tick-borne diseases</article-title>. <source>Sci Rep</source>. (<year>2018</year>) <volume>8</volume>:<fpage>15932</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-018-34393-9</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mosel</surname> <given-names>MR</given-names>
</name>
<name>
<surname>Carolan</surname> <given-names>HE</given-names>
</name>
<name>
<surname>Rebman</surname> <given-names>AW</given-names>
</name>
<name>
<surname>Castro</surname> <given-names>S</given-names>
</name>
<name>
<surname>Massire</surname> <given-names>C</given-names>
</name>
<name>
<surname>Ecker</surname> <given-names>DJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Molecular Testing of Serial Blood Specimens from Patients with Early Lyme Disease during Treatment Reveals Changing Coinfection with Mixtures of Borrelia burgdorferi Genotypes</article-title>. <source>Antimicrob Agents Chemother</source>. (<year>2019</year>) <volume>63</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/AAC.00237-19</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Seinost</surname> <given-names>G</given-names>
</name>
<name>
<surname>Golde</surname> <given-names>WT</given-names>
</name>
<name>
<surname>Berger</surname> <given-names>BW</given-names>
</name>
<name>
<surname>Dunn</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>D</given-names>
</name>
<name>
<surname>Dunkin</surname> <given-names>DS</given-names>
</name>
<etal/>
</person-group>. <article-title>Infection with multiple strains of Borrelia burgdorferi sensu stricto in patients with Lyme disease</article-title>. <source>Arch Dermatol</source>. (<year>1999</year>) <volume>135</volume>:<page-range>1329&#x2013;33</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/archderm.135.11.1329</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stanek</surname> <given-names>G</given-names>
</name>
<name>
<surname>Reiter</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>The expanding Lyme Borrelia complex&#x2013;clinical significance of genomic species</article-title>? <source>Clin Microbiol Infect</source>. (<year>2011</year>) <volume>17</volume>:<page-range>487&#x2013;93</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1469-0691.2011.03492.x</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cutler</surname> <given-names>S</given-names>
</name>
<name>
<surname>Vayssier-Taussat</surname> <given-names>M</given-names>
</name>
<name>
<surname>Estrada-Pe&#xf1;a</surname> <given-names>A</given-names>
</name>
<name>
<surname>Potkonjak</surname> <given-names>A</given-names>
</name>
<name>
<surname>Mihalca</surname> <given-names>AD</given-names>
</name>
<name>
<surname>Zeller</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>A new Borrelia on the block: Borrelia miyamotoi - a human health risk</article-title>? <source>Euro Surveill</source>. (<year>2019</year>) <volume>24</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.2807/1560-7917.Es.2019.24.18.1800170</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pritt</surname> <given-names>BS</given-names>
</name>
<name>
<surname>Respicio-Kingry</surname> <given-names>LB</given-names>
</name>
<name>
<surname>Sloan</surname> <given-names>LM</given-names>
</name>
<name>
<surname>Schriefer</surname> <given-names>ME</given-names>
</name>
<name>
<surname>Replogle</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Bjork</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Borrelia mayonii sp. nov., a member of the Borrelia burgdorferi sensu lato complex, detected in patients and ticks in the upper midwestern United States</article-title>. <source>Int J Syst Evol Microbiol</source>. (<year>2016</year>) <volume>66</volume>:<page-range>4878&#x2013;80</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1099/ijsem.0.001445</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aguero-Rosenfeld</surname> <given-names>ME</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>G</given-names>
</name>
<name>
<surname>Schwartz</surname> <given-names>I</given-names>
</name>
<name>
<surname>Wormser</surname> <given-names>GP</given-names>
</name>
</person-group>. <article-title>Diagnosis of lyme borreliosis</article-title>. <source>Clin Microbiol Rev</source>. (<year>2005</year>) <volume>18</volume>:<fpage>484</fpage>&#x2013;<lpage>509</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/CMR.18.3.484-509.2005</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arvey</surname> <given-names>A</given-names>
</name>
<name>
<surname>Rowe</surname> <given-names>M</given-names>
</name>
<name>
<surname>Legutki</surname> <given-names>JB</given-names>
</name>
<name>
<surname>An</surname> <given-names>G</given-names>
</name>
<name>
<surname>Gollapudi</surname> <given-names>A</given-names>
</name>
<name>
<surname>Lei</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Age-associated changes in the circulating human antibody repertoire are upregulated in autoimmunity</article-title>. <source>Immun Ageing</source>. (<year>2020</year>) <volume>17</volume>:<fpage>28</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12979-020-00193-x</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taguchi</surname> <given-names>AT</given-names>
</name>
<name>
<surname>Boyd</surname> <given-names>J</given-names>
</name>
<name>
<surname>Diehnelt</surname> <given-names>CW</given-names>
</name>
<name>
<surname>Legutki</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>ZG</given-names>
</name>
<name>
<surname>Woodbury</surname> <given-names>NW</given-names>
</name>
<etal/>
</person-group>. <article-title>Comprehensive prediction of molecular recognition in a combinatorial chemical space using machine learning</article-title>. <source>ACS Comb Sci</source>. (<year>2020</year>) <volume>35</volume>:<page-range>500&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acscombsci.0c00003</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kelbauskas</surname> <given-names>L</given-names>
</name>
<name>
<surname>Legutki</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Woodbury</surname> <given-names>NW</given-names>
</name>
</person-group>. <article-title>Highly heterogenous humoral immune response in Lyme disease patients revealed by broad machine learning-assisted antibody binding profiling with random peptide arrays</article-title>. <source>Front Immunol</source>. (<year>2024</year>) <volume>15</volume>:<elocation-id>1335446</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2024.1335446</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rowe</surname> <given-names>M</given-names>
</name>
<name>
<surname>Melnick</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gerwien</surname> <given-names>R</given-names>
</name>
<name>
<surname>Legutki</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Pfeilsticker</surname> <given-names>J</given-names>
</name>
<name>
<surname>Tarasow</surname> <given-names>TM</given-names>
</name>
<etal/>
</person-group>. <article-title>An ImmunoSignature test distinguishes Trypanosoma cruzi, hepatitis B, hepatitis C and West Nile virus seropositivity among asymptomatic blood donors</article-title>. <source>PloS Negl Trop Dis</source>. (<year>2017</year>) <volume>11</volume>:<elocation-id>e0005882</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pntd.0005882</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>L.</surname> <given-names>C</given-names>
</name>
<name>
<surname>Fiorentino</surname> <given-names>D</given-names>
</name>
<name>
<surname>Gerwien</surname> <given-names>R</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>K</given-names>
</name>
<name>
<surname>Legutki</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Vergara</surname> <given-names>AV</given-names>
</name>
<etal/>
</person-group>. <article-title>Immunosignature Technology Differentiates Patients with Systemic Sclerosis and Internal Organ Involvement</article-title>. In: <source>2016 ACR/ARHP Annual Meeting</source>. <publisher-loc>Washington, D.C</publisher-loc>: <publisher-name>American College of Rheumatology</publisher-name> (<year>2016</year>).</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sykes</surname> <given-names>KF</given-names>
</name>
<name>
<surname>Legutki</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Stafford</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>Immunosignaturing: a critical review</article-title>. <source>Trends Biotechnol</source>. (<year>2013</year>) <volume>31</volume>:<fpage>45</fpage>&#x2013;<lpage>51</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.tibtech.2012.10.012</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Legutki</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>ZG</given-names>
</name>
<name>
<surname>Greving</surname> <given-names>M</given-names>
</name>
<name>
<surname>Woodbury</surname> <given-names>N</given-names>
</name>
<name>
<surname>Johnston</surname> <given-names>SA</given-names>
</name>
<name>
<surname>Stafford</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>Scalable high-density peptide arrays for comprehensive health monitoring</article-title>. <source>Nat Commun</source>. (<year>2014</year>) <volume>5</volume>:<fpage>4785</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ncomms5785</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Putterman</surname> <given-names>C</given-names>
</name>
<name>
<surname>Rowe</surname> <given-names>M</given-names>
</name>
<name>
<surname>Legutki</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Tarasow</surname> <given-names>TM</given-names>
</name>
<name>
<surname>Sykes</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>A Simple Test for Assessing and Monitoring SLE Disease Activity Status</article-title>. In: <source>2016 ACR/ARHP Annual Meeting</source>. <publisher-loc>Washington, D.C</publisher-loc>: <publisher-name>American College of Rheumatology</publisher-name> (<year>2016</year>).</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bisarad</surname> <given-names>P</given-names>
</name>
<name>
<surname>Kelbauskas</surname> <given-names>L</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>A</given-names>
</name>
<name>
<surname>Taguchi</surname> <given-names>AT</given-names>
</name>
<name>
<surname>Trenchevska</surname> <given-names>O</given-names>
</name>
<name>
<surname>Woodbury</surname> <given-names>NW</given-names>
</name>
<etal/>
</person-group>. <article-title>Predicting monoclonal antibody binding sequences from a sparse sampling of all possible sequences</article-title>. <source>Commun Biol</source>. (<year>2024</year>) <volume>7</volume>:<fpage>979</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s42003-024-06650-3</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chowdhury</surname> <given-names>R</given-names>
</name>
<name>
<surname>Taguchi</surname> <given-names>AT</given-names>
</name>
<name>
<surname>Kelbauskas</surname> <given-names>L</given-names>
</name>
<name>
<surname>Stafford</surname> <given-names>P</given-names>
</name>
<name>
<surname>Diehnelt</surname> <given-names>C</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>ZG</given-names>
</name>
<etal/>
</person-group>. <article-title>Modeling the sequence dependence of differential antibody binding in the immune response to infectious disease</article-title>. <source>PloS Comput Biol</source>. (<year>2023</year>) <volume>19</volume>:<elocation-id>e1010773</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pcbi.1010773</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kirpach</surname> <given-names>J</given-names>
</name>
<name>
<surname>Colone</surname> <given-names>A</given-names>
</name>
<name>
<surname>B&#xfc;rckert</surname> <given-names>JP</given-names>
</name>
<name>
<surname>Faison</surname> <given-names>WJ</given-names>
</name>
<name>
<surname>Dubois</surname> <given-names>ARSX</given-names>
</name>
<name>
<surname>Sinner</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Detection of a low level and heterogeneous B cell immune response in peripheral blood of acute borreliosis patients with high throughput sequencing</article-title>. <source>Front Immunol</source>. (<year>2019</year>) <volume>10</volume>:<elocation-id>1105</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fimmu.2019.01105</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tibshirani</surname> <given-names>R</given-names>
</name>
<name>
<surname>Hastie</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>Outlier sums for differential gene expression analysis</article-title>. <source>Biostatistics</source>. (<year>2007</year>) <volume>8</volume>:<fpage>2</fpage>&#x2013;<lpage>8</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/biostatistics/kxl005</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Magnarelli</surname> <given-names>LA</given-names>
</name>    <name>
<surname>Ijdo</surname> <given-names>JW</given-names>
</name>
<name>
<surname>Padula</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Flavell</surname> <given-names>RA</given-names>
</name>
<name>
<surname>Fikrig</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>Serologic diagnosis of lyme borreliosis by using enzyme-linked immunosorbent assays with recombinant antigens</article-title>. <source>J Clin Microbiol</source>. (<year>2000</year>) <volume>38</volume>:<page-range>1735&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/JCM.38.5.1735-1739.2000</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tjernberg</surname> <given-names>I</given-names>
</name>
<name>
<surname>Kr&#xfc;ger</surname> <given-names>G</given-names>
</name>
<name>
<surname>Eliasson</surname> <given-names>I</given-names>
</name>
</person-group>. <article-title>C6 peptide ELISA test in the serodiagnosis of Lyme borreliosis in Sweden</article-title>. <source>Eur J Clin Microbiol Infect Dis</source>. (<year>2007</year>) <volume>26</volume>:<fpage>37</fpage>&#x2013;<lpage>42</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10096-006-0239-3</pub-id>
</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gr&#x105;&#x17a;lewska</surname> <given-names>W</given-names>
</name>
<name>
<surname>Holec-G&#x105;sior</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Antibody cross-reactivity in serodiagnosis of lyme disease</article-title>. <source>Antibodies (Basel)</source>. (<year>2023</year>) <volume>12</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/antib12040063</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>