<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1663484</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Adaptive noise-augmented attention for enhancing Transformer fine-tuning on longitudinal medical data</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Amirahmadi</surname> <given-names>Ali</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3111505/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Etminani</surname> <given-names>Farzaneh</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ohlsson</surname> <given-names>Mattias</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Center for Applied Intelligent Systems Research in Health, Halmstad University</institution>, <addr-line>Halmstad</addr-line>, <country>Sweden</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Research and Development (FoU), Region Halland</institution>, <addr-line>Halmstad</addr-line>, <country>Sweden</country></aff>
<aff id="aff3"><sup>3</sup><institution>Centre for Environmental and Climate Science, Computational Science for Health and Environment, Lund University</institution>, <addr-line>Lund</addr-line>, <country>Sweden</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Mini Han Wang, Zhuhai People&#x00027;s Hospital, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Antonio Sarasa-Cabezuelo, Complutense University of Madrid, Spain</p>
<p>Hunter Scarborough, John Peter Smith Hospital, United States</p>
<p>Hojjat Karami, Swiss Federal Institute of Technology Lausanne, Switzerland</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Ali Amirahmadi <email>ali.amirahmadi&#x00040;hh.se</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1663484</elocation-id>
<history>
<date date-type="received">
<day>10</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>25</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Amirahmadi, Etminani and Ohlsson.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Amirahmadi, Etminani and Ohlsson</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Transformer models pre-trained on self-supervised tasks and fine-tuned on downstream objectives have achieved remarkable results across a variety of domains. However, fine-tuning these models for clinical predictions from longitudinal medical data, such as electronic health records (EHR), remains challenging due to limited labeled data and the complex, event-driven nature of medical sequences. While self-attention mechanisms are powerful for capturing relationships within sequences, they may underperform when modeling subtle dependencies between sparse clinical events under limited supervision. We introduce a simple yet effective fine-tuning technique, Adaptive Noise-Augmented Attention (ANAA), which injects adaptive noise directly into the self-attention weights and applies a 2D Gaussian kernel to smooth the resulting attention maps. This mechanism broadens the attention distribution across tokens while refining it to emphasize more informative events. Unlike prior approaches that require expensive modifications to the architecture and pre-training phase, ANAA operates entirely during fine-tuning. Empirical results across multiple clinical prediction tasks demonstrate consistent performance improvements. Furthermore, we analyze how ANAA shapes the learned attention behavior, offering interpretable insights into the model&#x00027;s handling of temporal dependencies in EHR data.</p></abstract>
<kwd-group>
<kwd>Transformer</kwd>
<kwd>augmentation</kwd>
<kwd>adaptive noise</kwd>
<kwd>medical data</kwd>
<kwd>electronic health records (EHR)</kwd>
<kwd>fine-tuning</kwd>
<kwd>representation learning</kwd>
<kwd>self-attention</kwd>
</kwd-group>
<contract-sponsor id="cn001">Vetenskapsr&#x000E5;det<named-content content-type="fundref-id">https://doi.org/10.13039/501100004359</named-content></contract-sponsor>
<contract-sponsor id="cn002">Stiftelsen f&#x000F6;r Kunskaps- och Kompetensutveckling<named-content content-type="fundref-id">https://doi.org/10.13039/501100003170</named-content></contract-sponsor>
<counts>
<fig-count count="6"/>
<table-count count="3"/>
<equation-count count="12"/>
<ref-count count="51"/>
<page-count count="12"/>
<word-count count="8481"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Medicine and Public Health</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Foundation models, deep neural networks pre-trained on broad unlabeled data using self-supervised methods, have significantly impacted various aspects of our lives, including law, healthcare, education, and more (<xref ref-type="bibr" rid="B6">Bommasani et al., 2021</xref>; <xref ref-type="bibr" rid="B17">Guo et al., 2023</xref>; <xref ref-type="bibr" rid="B44">Wornow et al., 2023</xref>). These models typically acquire general knowledge about the data through pre-training a variant of the Transformer network on a self-supervised task like Masked Language Model (MLM), and then adapt this knowledge to downstream tasks with only a few labeled samples during the fine-tuning process. Researchers showed that pre-training, even with limited data, can improve Transformers&#x00027; performance significantly (<xref ref-type="bibr" rid="B3">Amos et al., 2023</xref>).</p>
<p>Pre-training Transformers have been employed with various self-supervised objectives and domains. Common objectives include corrupted text reconstruction tasks like MLM (<xref ref-type="bibr" rid="B13">Devlin et al., 2018</xref>; <xref ref-type="bibr" rid="B28">Lewis et al., 2019</xref>; <xref ref-type="bibr" rid="B27">Lan et al., 2019</xref>) and standard language models such as next-word prediction (<xref ref-type="bibr" rid="B36">Radford et al., 2019</xref>; <xref ref-type="bibr" rid="B7">Brown et al., 2020</xref>), which have been extensively utilized (<xref ref-type="bibr" rid="B32">Liu et al., 2023</xref>). These models typically adopt a backbone architecture inspired by the multi-head attention mechanism in Transformers (<xref ref-type="bibr" rid="B43">Vaswani et al., 2017</xref>), known for its effectiveness in modeling complex interaction between events (tokens) in a sequence (text). These foundation models have been pre-trained on different domain data (<xref ref-type="bibr" rid="B27">Lan et al., 2019</xref>; <xref ref-type="bibr" rid="B36">Radford et al., 2019</xref>), including structured temporal health data as sequences of events (<xref ref-type="bibr" rid="B31">Li et al., 2020</xref>; <xref ref-type="bibr" rid="B37">Rasmy et al., 2021</xref>; <xref ref-type="bibr" rid="B34">Pang et al., 2021</xref>).</p>
<p>Modeling Electronic Health Records (EHRs) trajectories presents a critical opportunity for predicting health-related outcomes, offering benefits like early intervention, cost reduction, and improved public health. This field has attracted significant attention from deep learning researchers (<xref ref-type="bibr" rid="B46">Xiao et al., 2018</xref>; <xref ref-type="bibr" rid="B2">Amirahmadi et al., 2023</xref>; <xref ref-type="bibr" rid="B5">Boll et al., 2024</xref>; <xref ref-type="bibr" rid="B29">Li et al., 2024</xref>). Typically, healthcare specific foundation models are pre-trained on publicly available, unlabeled EHR data, and adapting these models through fine-tuning consistently demonstrates superior performance across various tasks (<xref ref-type="bibr" rid="B31">Li et al., 2020</xref>; <xref ref-type="bibr" rid="B37">Rasmy et al., 2021</xref>; <xref ref-type="bibr" rid="B34">Pang et al., 2021</xref>; <xref ref-type="bibr" rid="B38">Ren et al., 2021</xref>; <xref ref-type="bibr" rid="B30">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B48">Yuanyuan et al., 2025</xref>).</p>
<p>However, EHRs are often scarce, and training Transformers to learn the complex relationships between medical events in longitudinal EHRs requires either large amounts of data, or advanced training techniques and augmentations (<xref ref-type="bibr" rid="B15">Dosovitskiy et al., 2020</xref>; <xref ref-type="bibr" rid="B42">Touvron et al., 2021</xref>; <xref ref-type="bibr" rid="B20">Hassani et al., 2021</xref>, <xref ref-type="bibr" rid="B19">2023</xref>). Due to privacy concerns and the scarcity of publicly available datasets, models often fail to learn the intricate dependencies between events in a patient&#x00027;s history. To address this, (<xref ref-type="bibr" rid="B11">Choi et al. 2020</xref>) proposed incorporating domain knowledge into the attention mechanism, while (<xref ref-type="bibr" rid="B51">Zhu and Razavian 2021</xref>) employed variational regularization. Additionally, (<xref ref-type="bibr" rid="B1">Amirahmadi et al. 2025</xref>) suggested pre-training the Transformer on the MLM task and the ordering of medical events in a patient&#x00027;s history, and (<xref ref-type="bibr" rid="B24">Kim and Lee 2024</xref>) proposed using learnable, adaptive kernels in the attention matrices to improve contextual representations and enhance the learned structure through self-attention. <xref ref-type="fig" rid="F1">Figures 1</xref>, <xref ref-type="fig" rid="F4">4</xref> illustrate how these various approaches impact self-attention behaviors in leaning the relationships between events. However, these methods often come with substantial computational costs and require extra effort for implementation and design.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Visualization of attention score patterns for different models from previous studies and how their proposed methods helping a more complicated structure in attention scores in Transformers. <bold>(a, b)</bold> Transformer trained from random weights vs. Transformer trained with domain knowledge (<xref ref-type="bibr" rid="B51">Zhu and Razavian, 2021</xref>; <xref ref-type="bibr" rid="B11">Choi et al., 2020</xref>). <bold>(c, d)</bold> Encoder-decoder vs. VGNN using variational regularization (<xref ref-type="bibr" rid="B51">Zhu and Razavian, 2021</xref>). <bold>(e, f)</bold> Vanilla Transformer vs. SAT with temporal priors (<xref ref-type="bibr" rid="B24">Kim and Lee, 2024</xref>). <bold>(g, h)</bold> Transformer pre-trained on MLM vs. MLM with trajectory order prediction (<xref ref-type="bibr" rid="B1">Amirahmadi et al., 2025</xref>). <bold>(e, f)</bold> Had no color bars in the original papers.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1663484-g0001.tif">
<alt-text>Eight heatmaps comparing various models: (a) Transformer with distinct blocks; (b) GCT with guided regularization, showing diagonal patterns; (c) Enc-dec Transformer, with varied blocks; (d) VGNN Transformer, with scattered patterns; (e) Transformer with smooth gradient; (f) SAT, showing a gradient shift; (g) Pre-trained Transformer, featuring patterns and gradients; (h) TOO-BERT, with distinct mixed shades. Each displays specific pattern variations.</alt-text>
</graphic>
</fig>
<p>Data augmentation is another solution to tackle the data scarcity challenge. Augmenting data with discrete data types, such as series of medical codes or tokens in text, is challenging because small perturbations can drastically alter semantic meaning, and interpolation in discrete space is not feasible (<xref ref-type="bibr" rid="B9">Chen et al., 2020</xref>). For example, replacing a code for &#x0201C;Type 1 diabetes&#x0201D; with &#x0201C;Type 2 diabetes,&#x0201D; or reordering diagnosis and procedure codes within the same patient trajectory, can fundamentally change the clinical context. As a result, researchers have proposed augmenting models during training as an alternative (<xref ref-type="bibr" rid="B21">Jain et al., 2023</xref>; <xref ref-type="bibr" rid="B49">Zehui et al., 2019</xref>; <xref ref-type="bibr" rid="B45">Wu et al., 2023</xref>).</p>
<p>In this study, we propose a simple two-step augmentation technique-Adaptive Noise-Augmented Attention (ANAA)&#x02014;that perturbs attention scores by injecting adaptive Gaussian noise followed by smoothing with a Gaussian kernel. Our investigation of attention distributions reveals that fine-tuned Transformers tend to produce highly polarized attention scores&#x02014;values clustering near the extremes (0 or 1), which restricts the model&#x00027;s capacity to explore diverse dependencies (see the bottom row of <bold>Figure 4</bold>). By introducing controlled noise into attention scores during fine-tuning, we encourage exploration of alternative dependency paths between events. The subsequent smoothing operation helps restore structural consistency while preserving diversity, resulting in more balanced and informative self-attention maps.</p>
<p>The main contributions are summarized as follows:</p>
<list list-type="order">
<list-item><p>We proposed a simple self-attention augmentation method that encourages the model to explore and learn more complex attention patterns during fine-tuning. Importantly, this approach does not modify the computational graph, making it easily applicable to any pre-trained Transformer.</p></list-item>
<list-item><p>We conducted several evaluations on various downstream tasks, examining the effect of the novel method on model performance, model robustness with limited training samples, and the balance of attention distribution between distant and nearby events. Our results demonstrate how it improves the performance of pre-trained Transformers.</p></list-item>
</list></sec>
<sec id="s2">
<title>2 Preliminary</title>
<sec>
<title>2.1 Transformer encoder and self-attention</title>
<p>The core back-bone of Transformers encoder is the multi-head self-attention. Each self-attention head is:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>X</mml:mi><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>Q</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>X</mml:mi><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>X</mml:mi><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">softmax</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Self-attention</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where, <inline-formula><mml:math id="M4"><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="M5"><mml:mi>V</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> and <italic>n</italic> is the length of input sequence and <italic>d</italic><sub><italic>k</italic></sub> and <italic>d</italic><sub><italic>v</italic></sub> are dimenssion of Key and Value. <italic>A</italic><sub><italic>h</italic></sub> is the attention score matrix and each <italic>A</italic><sub><italic>i, j</italic></sub> indicates how much attention token <italic>x</italic><sub><italic>i</italic></sub> put on <italic>x</italic><sub><italic>j</italic></sub>. Transformer encoders, is built on concatenation of &#x02223;<italic>h</italic>&#x02223; number attention heads in parallel, so each one has its own weights. Then, the concatenation is projected:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">MultiHead</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Concat</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02223;</mml:mo><mml:mi>h</mml:mi><mml:mo>&#x02223;</mml:mo></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>O</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where, <inline-formula><mml:math id="M7"><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>O</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02223;</mml:mo><mml:mi>h</mml:mi><mml:mo>&#x02223;</mml:mo><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> Multiple self-attention heads in parallel, help the model to attend to information from different representation subspaces (<xref ref-type="bibr" rid="B43">Vaswani et al., 2017</xref>; <xref ref-type="bibr" rid="B18">Hao et al., 2021</xref>).</p>
</sec>
<sec>
<title>2.2 Pre-training, fine-tuning</title>
<p>Pretraining typically involves the model acquiring general knowledge, which is then used to initialize the final network. Subsequently, the final network adjusts these weights to obtain optimized weights for specific downstream tasks (<xref ref-type="bibr" rid="B10">Chen et al., 2021</xref>). This approach has been extensively utilized for adapting foundation models to downstream tasks (<xref ref-type="bibr" rid="B27">Lan et al., 2019</xref>; <xref ref-type="bibr" rid="B32">Liu et al., 2023</xref>).</p>
</sec>
</sec>
<sec id="s3">
<title>3 Related works</title>
<p>Advanced training techniques and data augmentation have been widely adopted to improve the performance of Transformer models, especially in settings with limited labeled data. These methods aim to enhance the generalizability and robustness of learned representations.</p>
<p>Several methods modify self-attention to better learn intricate local and global attentions between different tokens. (<xref ref-type="bibr" rid="B19">Hassani et al. 2023</xref>) introduced a sliding window attention mechanism to localize attention spans and improve efficiency. (<xref ref-type="bibr" rid="B14">Ding et al. 2023</xref>) reduced attention complexity by segmenting key, query, and value inputs and sparsifying their interactions, allowing Transformers to better model both short- and long-range dependencies. Positional encoding has also been a target for improvement: (<xref ref-type="bibr" rid="B40">Su et al. 2024</xref>) and <xref ref-type="bibr" rid="B35">Press et al. (2021</xref>) enhanced distant token interaction by encoding absolute positions with rotation matrices or distance-based penalties on query-key attention scores. While these methods are effective, they often require structural changes to the attention mechanism, making them less compatible with pre-trained models and harder to integrate into existing pipelines.</p>
<p>Data augmentation is another solution to tackle the data scarcity challenge, but it is particularly challenging in discrete domains like medical codes or text, where small changes can drastically alter semantic meaning and interpolation is not well-defined (<xref ref-type="bibr" rid="B9">Chen et al., 2020</xref>). To address this, researchers have proposed augmenting models during training or fine-tuning by injecting noise into internal representations (<xref ref-type="bibr" rid="B21">Jain et al., 2023</xref>; <xref ref-type="bibr" rid="B49">Zehui et al., 2019</xref>; <xref ref-type="bibr" rid="B47">Yuan et al., 2022</xref>; <xref ref-type="bibr" rid="B44">Wornow et al., 2023</xref>; <xref ref-type="bibr" rid="B45">Wu et al., 2023</xref>). Injecting Gaussian noise into activations has been shown to help models converge to smoother minima, improving generalization, calibration, and robustness to perturbations (<xref ref-type="bibr" rid="B8">Camuto et al., 2020</xref>). (<xref ref-type="bibr" rid="B50">Zhu et al. 2019</xref>) enhanced the performance of BERT (<xref ref-type="bibr" rid="B13">Devlin et al., 2018</xref>) and RoBERTa (<xref ref-type="bibr" rid="B33">Liu et al., 2019</xref>) by adding adversarial noise to word embeddings, a technique later extended to graph neural networks by (<xref ref-type="bibr" rid="B25">Kong et al. 2022</xref>) for improved out-of-distribution generalization. In the self-attention space, (<xref ref-type="bibr" rid="B49">Zehui et al. 2019</xref>) proposed DropAttention, which randomly masks and expands attention scores to regularize focus. Similarly, (<xref ref-type="bibr" rid="B45">Wu et al. 2023</xref>) introduced adversarial structural biases to attention matrices, though at the cost of increased training complexity.</p>
<p>(<xref ref-type="bibr" rid="B44">Wornow et al. 2023</xref>) injected Gaussian noise into the latent space of an encoder-decoder model for better image captioning, while (<xref ref-type="bibr" rid="B47">Yuan et al. 2022</xref>) perturbed hidden representations during fine-tuning to marginally improve language model performance. Most notably, (<xref ref-type="bibr" rid="B21">Jain et al. 2023</xref>) introduced NEFTune, which adds calibrated uniform noise to embedding vectors during fine-tuning&#x02014;resulting in significant improvements for models like LLaMA-1 and LLaMA-2. Inspired by these efforts, we compare our method with NEFTune and propose a new approach that directly perturbs the attention scores, encouraging the model to learn richer contextual dependencies across sequences. Here, We investigate augmenting the self-attention scores&#x02014;central to modeling event dependencies&#x02014;by injecting and smoothing adaptive Gaussian noise. Unlike prior methods that perturb embeddings or hidden states, our approach directly improves attention behavior without changing the model architecture, enhancing the learned representation in a lightweight and effective way.</p>
</sec>
<sec sec-type="methods" id="s4">
<title>4 Methods</title>
<sec>
<title>4.1 Adaptive noise-augmented attention</title>
<p>In this subsection, we introduce, Adaptive Noise-Augmented Attention (ANAA), a simple yet effective two-step augmentation technique designed to improve the learned representations in Transformer models by directly augmenting the attention scores during fine-tuning (<xref ref-type="table" rid="T4">Algorithm 1</xref>). This method enhances attention dynamics without modifying the computational graph, making it compatible with any pre-trained Transformer encoder.</p>
<table-wrap position="float" id="T4">
<label>Algorithm 1</label>
<caption><p>Fine-tuning Transformer encoder with ANAA.</p></caption>
<table frame="box" rules="all">
<tbody>
<tr>
<td valign="top" align="left"><bold>Input</bold>: <inline-formula><mml:math id="M8"><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">fine-tuning</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> tokenized dataset, embedding layer emb(&#x000B7;), attention score matrix <italic>A</italic><sub><italic>h</italic></sub>, normal noise <inline-formula><mml:math id="M9"><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003BC;</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">GN</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, two-dimensional Gaussian noise <italic>n</italic><sub>&#x003C3;<sub>eh</sub></sub>, rest of the model <italic>f</italic>(&#x000B7;) </td>
</tr>
<tr>
<td valign="top" align="left"><bold>Parameter</bold>: Normal noise <inline-formula><mml:math id="M10"><mml:mi>&#x003BC;</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">GN</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> calculated from <italic>A</italic><sub><italic>h</italic></sub>, event horizon hyperparameter &#x003C3;<sub>eh</sub> based on the data charecterstic needs to adjust the smoothing noise</td>
</tr>
<tr>
<td valign="top" align="left">1Initialize &#x003B8; from a pre-trained model 2<bold>repeat</bold></td>
</tr>
<tr>
<td valign="top" align="left">3 Sample (<italic>X</italic><sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub>)&#x0007E;<italic>D</italic><sub>fine-tuning</sub> <italic>X</italic><sub>emb</sub>&#x02190;emb(<italic>X</italic><sub><italic>i</italic></sub>)</td>
</tr>
<tr>
<td valign="top" align="left"><bold>for</bold> each Attention Head <italic>A</italic><sub><italic>h</italic></sub> in Transformer Block</td>
</tr>
<tr>
<td valign="top" align="left"><bold>do</bold> <inline-formula><mml:math id="M11"><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">attn</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02190;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">emb</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003BC;</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">GN</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="top" align="left"><italic>A</italic><sub><italic>h</italic></sub>(<italic>X</italic><sub>attn</sub>)&#x02190;Convolve(<italic>A</italic><sub><italic>h</italic></sub>(<italic>X</italic><sub>attn</sub>), <italic>n</italic><sub>&#x003C3;<sub>eh</sub></sub>) <italic>H</italic><sub><italic>h</italic></sub>(<italic>X</italic><sub>attn</sub>)&#x02190;<italic>A</italic><sub><italic>h</italic></sub>(<italic>X</italic><sub>attn</sub>)<italic>V</italic> <bold>end for</bold> MultiHead(<italic>H</italic>)&#x02190;concat(<italic>H</italic><sub>0</sub>(<italic>X</italic><sub>attn</sub>), &#x02026;, <italic>H</italic><sub><italic>h</italic></sub>(<italic>X</italic><sub>attn</sub>)) &#x00177;<sub><italic>i</italic></sub>&#x02190;<italic>f</italic>(MultiHead(<italic>H</italic>)) &#x003B8;&#x02190;opt(&#x003B8;, loss(&#x00177;<sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub>)) <bold>until</bold> Stopping criteria met or maximum iterations reached</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>ANAA operates by first injecting adaptive Gaussian noise into the attention score matrix and then applying a smoothing operation using a Gaussian kernel. This process encourages the model to explore the attention patterns and strengthens context modeling. The augmented attention is computed as:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">ANAA</mml:mtext><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mo>&#x0007E;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003BC;</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>*</mml:mo><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>V</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here, the Gaussian noise <inline-formula><mml:math id="M13"><mml:mrow><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003BC;</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">GN</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> is computed adaptivly based on the learned attention during training:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>&#x003BC;</mml:mi></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E7"><label>(7)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">GN</mml:mtext></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:msqrt></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The smoothing kernel <italic>n</italic><sub>&#x003C3;<sub>eh</sub></sub>[<italic>i, j</italic>] is a 2D Gaussian distribution:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">eh</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>&#x003C0;</mml:mi><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">eh</mml:mtext></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">eh</mml:mtext></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003C3;<sub>eh</sub> is a tunable hyperparameter representing the event horizon, controlling and adjusting the extent of the smoothing. The convolution operation &#x0002A; applies this kernel over the noise-augmented attention matrix:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>*</mml:mo><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>&#x003C0;</mml:mi><mml:msup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mi>f</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>k</italic> &#x0003D; 2&#x003C0;&#x003C3; is the kernel size.</p>
<p>This smoothing step modulates the added noise, reinforcing stronger attention patterns while allowing for broader exploration in attentions space. The noise parameters &#x003BC; and &#x003C3;<sub>GN</sub> are computed independently for each attention head to preserve head-specific attention dynamics during training. <xref ref-type="fig" rid="F2">Figure 2</xref> illustrates the full ANAA mechanism.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Adaptive Noise-Augmented Attention (ANAA) mechanism.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1663484-g0002.tif">
<alt-text>Diagram illustrating a mathematical operation involving matrices and functions. A green matrix labeled Q is multiplied by a blue transposed matrix K, followed by the addition of noise with a mean and standard deviation. This result undergoes a SoftMax function, then multiplied by a grayscale square, and finally applied to an orange matrix labeled V. </alt-text>
</graphic>
</fig>
<p>Adding adaptive Gaussian noise <inline-formula><mml:math id="M18"><mml:mo>&#x0007E;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003BC;</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> to the attention scores helps the model escape sub-optimal solutions and promotes learning more diverse interactions between events. The subsequent Gaussian convolution adjusts the magnitude and distribution of the injected noise, encouraging the model to focus on more meaningful and effective attention patterns.</p>
<p>During inference, stochasticity from the added noise is removed by replacing it with its expected value &#x003BC;, ensuring deterministic predictions:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">ANAA</mml:mtext><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>*</mml:mo><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>V</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The computational complexity of ANAA is <italic>O</italic>(<italic>n</italic><sup>2</sup>) (for more details, see the <xref ref-type="supplementary-material" rid="SM1">Supplementary material Section 1.9</xref>, and since it&#x00027;s primarily used during fine-tuning with limited labeled samples, the additional cost is negligible.</p>
</sec>
<sec>
<title>4.2 Mechanistic rationale: Why ANAA works</title>
<p><bold>Figure 4</bold> reveals that, in the absence of augmentation, many attention heads converge to a degenerate two-point distribution: each weight is either exactly 0 (&#x0201C;off&#x0201D;) or 1 (&#x0201C;on&#x0201D;), with masses 1&#x02212;&#x003B1; and &#x003B1;, respectively. ANAA first perturbs the Attention scores with adaptive Gaussian noise whose variance scales as &#x003B1;(1&#x02212;&#x003B1;) (<xref ref-type="supplementary-material" rid="SM1">Supplementary material Section 1.5</xref>). So, every Dirac spike is broadened into a narrow normal curve, turning the rigid on/off pattern into a bimodal continuous distribution that expresses graded less-important vs. more-important scores. The subsequent Gaussian convolution (<xref ref-type="supplementary-material" rid="SM1">Supplementary material Section 1.6</xref>) behaves as a data-adaptive low-pass filter: it suppresses high-frequency artifacts and interpolates between neighboring tokens, so the random, isolated spikes introduced by the noise disappear.</p>
<p>Taken together, ANAA can be viewed as a variance-scaled, structured drop-connect regularizer (<xref ref-type="supplementary-material" rid="SM1">Supplementary material Section 1.7</xref>), analogous to&#x02013;but more principled than&#x02013;classical dropout, which disconnects token pairs with an independent Bernoulli mask. ANAA instead perturbs each attention score additively, so every mini-batch sees a different, spatially smoothed view of the inputs relations.</p>
</sec>
</sec>
<sec id="s5">
<title>5 Experiments</title>
<sec>
<title>5.1 Datasets</title>
<p>In our study, we utilized medical data from two sources: the MIMIC-IV (<xref ref-type="bibr" rid="B23">Johnson et al., 2020</xref>) hosp module and the Malm&#x000F6; Diet and Cancer Cohort (MDC) (<xref ref-type="bibr" rid="B4">Berglund et al., 1993</xref>) dataset, approved by the Ethics Review Board of Sweden (Dnr 2023-00503-01). Each EHR trajectory represents a sequence of temporally structured health events. The MIMIC-IV dataset includes 173,000 patient records across 407,000 visits from 2008 to 2019, with 10.6 million medical codes. The MDC dataset, from a cohort study in Sweden, comprises 30,000 individuals with 531,000 visits from 1992 to 2020, offering a more extended patient history&#x02014;257 codes per patient on average, compared to MIMIC-IV&#x00027;s 61. To ensure consistency, we used only ICD and ATC codes, the only types available in MDC at the beginning, aligning with prior work like Med-BERT on diagnosis codes for risk prediction.</p>
<p>Both datasets use ICD and ATC codes for disease and medication classification. We randomly split each cohort into 70% for pre-training, 20% for fine-tuning, and 10% for testing. After preprocessing, MIMIC-IV had 2,195 unique ICD-9 and 137 ATC-5 codes, while MDC had 1,558 ICD-10 and 111 ATC-5 codes. To assess the generalizability and robustness of our results, the fine-tuning dataset was split into 5 folds. The model was fine-tuned on 4 folds with early stopping on the remaining fold, repeated 5 times with different validation sets. We reported the mean and standard deviation of the AUC on the unseen test dataset. For details, refer to the dataset availability, specifications and implementation details in the <xref ref-type="supplementary-material" rid="SM1">Supplementary material Sections 1.1</xref>, <xref ref-type="supplementary-material" rid="SM1">1.2</xref>, <xref ref-type="supplementary-material" rid="SM1">1.4</xref>.</p>
</sec>
<sec>
<title>5.2 Problem formulation</title>
<p>Each dataset <italic>D</italic> comprises a set of patients <italic>P</italic>, <italic>D</italic> &#x0003D; {<italic>P</italic><sup>1</sup>, <italic>P</italic><sup>2</sup>, &#x02026;, <italic>P</italic><sup>|<italic>D</italic>|</sup>}. In our study, we considered a total of |<italic>D</italic>| &#x0003D; 172, 980 patients for MIMIC-IV and |<italic>D</italic>| &#x0003D; 29, 664 patients for the MDC cohort. We represent each patient&#x00027;s longitudinal medical trajectory through a structured set of visit encounters as a sequence of events. This representation is denoted as <inline-formula><mml:math id="M20"><mml:mrow><mml:msup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, where <italic>O</italic> represents the total number of visit encounters for patient <italic>i</italic>. Each visit <inline-formula><mml:math id="M21"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0222A;</mml:mo><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> is the union of all diagnosis codes <italic>I</italic><sub><italic>j</italic></sub>&#x02282;<italic>I</italic> and prescribed medications <italic>M</italic><sub><italic>j</italic></sub>&#x02282;<italic>M</italic> that are recorded for the <italic>P</italic><sup><italic>i</italic></sup> at visit <inline-formula><mml:math id="M22"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>. To reduce sparsity, we excluded less frequently occurring medical codes and retained only the initial 4 digits of ICD and ATC codes.</p>
<p>To guide the model in understanding changes in encounter times and the structure of each patient&#x00027;s trajectory, similar to BERT, we employed special tokens. A [<italic>CLS</italic>] token is placed at the beginning of each patient&#x00027;s trajectory, while a [<italic>SEP</italic>] token is inserted between visits. Each visit represents a set of diagnoses and medications recorded within a specific time span, and the [<italic>SEP</italic>] token separates the sets of medical codes from one visit to the next. Consequently, each patient&#x00027;s trajectory is represented as <inline-formula><mml:math id="M23"><mml:mrow><mml:msup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>C</mml:mi><mml:mi>L</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, providing the model with valuable context for analysis and prediction.</p>
<p>Here, we evaluated our models on 3 downstream tasks <italic>e</italic><sub><italic>dt</italic></sub> [Heart Failure (HF), Alzheimer&#x00027;s Disease (AD), Prolonged Length of Stay on the next visit (PLS) predictions], where the model predicts the incidence of the first HF (<italic>I</italic><sub><italic>N</italic> &#x0003D; <italic>HF</italic></sub>) or AD (<italic>I</italic><sub><italic>N</italic> &#x0003D; <italic>AD</italic></sub>) ICD codes or the presence of PLS (<italic>PLS</italic><sub><italic>N</italic></sub> &#x0003D; 1) on the <italic>N</italic><sup><italic>th</italic></sup> visit, given the patient&#x00027;s previous history of medical codes, <inline-formula><mml:math id="M24"><mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>:</mml:mo><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, as a sequence of temporally structured health events:</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>&#x02119;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02223;</mml:mo><mml:msup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>C</mml:mi><mml:mi>L</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>For each patient&#x00027;s trajectory, if there were no occurrences of the target events <italic>e</italic><sub><italic>dt</italic></sub>, it is considered a negative case; otherwise, we exclude the first visit with the target and all subsequent visits and consider it a positive case. All ATC codes related to HF treatment are excluded to avoid timing-related noise and non-trivial predictions. Initially, models exhibited bias toward longer visit histories, confounding risk predictions. To address this, we excluded trajectories with fewer than 30 visits in the MDC dataset and fewer than 10 visits in the MIMIC-IV dataset. This ensured balanced visit histories between positive and negative cases, resulting in averages of 19 visits in the MDC dataset and 9 visits in the MIMIC-IV dataset, aligning with their overall dataset averages prior to preprocessing. <xref ref-type="table" rid="T1">Table 1</xref> summarizes the number of positive and negative cases after these preprocessing steps.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Number of positive and negative samples in each downstream task.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Task</bold></th>
<th valign="top" align="center"><bold>Positive</bold></th>
<th valign="top" align="center"><bold>Negative</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">PLS prediction</td>
<td valign="top" align="center">2,429</td>
<td valign="top" align="center">6,360</td>
</tr> <tr>
<td valign="top" align="left">HF prediction (MIMIC-IV)</td>
<td valign="top" align="center">243</td>
<td valign="top" align="center">641</td>
</tr> <tr>
<td valign="top" align="left">AD prediction</td>
<td valign="top" align="center">245</td>
<td valign="top" align="center">2,628</td>
</tr> <tr>
<td valign="top" align="left">HF prediction (MDC)</td>
<td valign="top" align="center">103</td>
<td valign="top" align="center">301</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>5.3 List of models</title>
<p>To thoroughly investigate the impact of the proposed ANAA augmentation, we compared the performance of following conventional and deep learning models on downstream tasks of HF, AD, and PLS prediction using both the MDC and MIMIC-IV datasets. These models were trained either from scratch or initiated from pre-trained weights, fine-tuned on the fine-tuning dataset, and evaluated on the test dataset. We set the tunable event horizon parameter to &#x003C3;<sub><italic>eh</italic></sub> &#x0003D; 1.0 (kernel size = 6) for the ANAA on the MDC dataset and &#x003C3;<sub><italic>eh</italic></sub> &#x0003D; 0.33 (kernel size = 2) on the MIMIC IV after fine-tuning on the fine-tuning dataset. Except fir HF prediction in the MDC, different &#x003C3;<sub><italic>eh</italic></sub>, slightly changes the ANAA performance. For more details see <xref ref-type="supplementary-material" rid="SM1">Supplementary material Section 1.3</xref>.</p>
<sec>
<title>5.3.1 Models with proposed RNA/ANAA</title>
<list list-type="bullet">
<list-item><p><bold>Transformer with ANAA</bold>: This model incorporates ANAA into all self-attention heads of a randomly initialized Transformer.</p></list-item>
<list-item><p><bold>Transformer pre-trained on MLM with Raw Noise injected Attention (RNA)</bold>: In this approach, <inline-formula><mml:math id="M26"><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003BC;</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> (normal noise with adaptive parameters) is added to all self-attention heads of a pre-trained Transformer. This experiment allows us to isolate the impact of the noise injection from the smoothing effect of Gaussian convolution.</p></list-item>
<list-item><p><bold>Transformer pre-trained on MLM with ANAA</bold>: This model incorporates ANAA into all self-attention heads of the pre-trained Transformer.</p></list-item>
</list>
<p>Baseline model details and results are provided in <xref ref-type="supplementary-material" rid="SM1">Supplementary material Section 1.11</xref>.</p>
</sec>
</sec>
<sec>
<title>5.4 Evaluation on downstream tasks</title>
<p>The results are summarized in <xref ref-type="table" rid="T2">Table 2</xref> and suggest that adding ANAA improves the AUC of pre-trained Transformers, potentially positioning them as one of the state-of-the-art methods for outcome prediction on temporal structured health data. Specifically, on the MDC dataset, the AUC for HF and AD prediction increased to 74.5% and 73.2%, respectively, while on the MIMIC-IV dataset, the AUC for HF prediction reached 87.2%. The addition of ANAA resulted in statistically significant improvements for HF prediction on both the MDC and MIMIC-IV datasets for the MLM pre-trained Transformer. Furthermore, the improvement in AD prediction was considerable, showcasing the effectiveness of ANAA augmentation. However, incorporating ANAA did not significantly alter the performance of PLS prediction. Additionally, applying ANAA to randomly initialized Transformers boosted the AUC for PLS prediction to 60.2%, with negligible effects on other downstream tasks. To delve deeper into the impact of each noise injection and smoothing augmentation term, we solely added the normal noise to the pre-trained Transformer. This experiment revealed that the noise injection alone had a more pronounced effect on downstream tasks in the MIMIC dataset, whereas the combined (ANAA) terms exhibited greater impacts on the downstream tasks in the MDC dataset, particularly associated with its longer sequences.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Average AUC values (%) and standard deviation for different methods for the HF prediction, AD prediction, and PLS prediction downstream tasks on the test datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model/downstream task</bold></th>
<th valign="top" align="center"><bold>HF prediction (MDC)</bold></th>
<th valign="top" align="center"><bold>AD prediction (MDC)</bold></th>
<th valign="top" align="center"><bold>HF prediction (MIMIC-IV)</bold></th>
<th valign="top" align="center"><bold>PLS prediction (MIMIC-IV)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="center">71.4 (0.5)</td>
<td valign="top" align="center">70.5 (0.8)</td>
<td valign="top" align="center">84.2 (1.4)</td>
<td valign="top" align="center">54.4 (0.8)</td>
</tr> <tr>
<td valign="top" align="left">Transformer&#x0002B; ANAA</td>
<td valign="top" align="center">72.1 (2.7)</td>
<td valign="top" align="center">70.4 (0.6)</td>
<td valign="top" align="center">83.2 (2.5)</td>
<td valign="top" align="center">60.2 (1.2)</td>
</tr> <tr>
<td valign="top" align="left">Transformer pre-trained on MLM</td>
<td valign="top" align="center">72.2 (2.5)</td>
<td valign="top" align="center">72.2 (1.1)</td>
<td valign="top" align="center">85.2 (1.1)</td>
<td valign="top" align="center">60.3 (1.3)</td>
</tr> <tr>
<td valign="top" align="left">Transformer pre-trained on MLM&#x0002B; RNA</td>
<td valign="top" align="center">72.6 (1.9)</td>
<td valign="top" align="center">71.4 (1.0)</td>
<td valign="top" align="center">86.5 (1.2)</td>
<td valign="top" align="center"><bold>60.7 (0.6)</bold></td>
</tr> <tr>
<td valign="top" align="left">Transformer pre-trained on MLM&#x0002B; ANAA</td>
<td valign="top" align="center"><bold>74.5 (2.9)</bold></td>
<td valign="top" align="center"><bold>73.2 (0.3)</bold></td>
<td valign="top" align="center"><bold>87.2 (0.4)</bold></td>
<td valign="top" align="center">60.3 (0.7)</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Boldface indicates the best-performing model.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>5.5 Performance boost on data insufficiency</title>
<p>One of the advantages of using pre-trained Transformers is their robustness and performance in situations of data insufficiency, observed in both NLP (<xref ref-type="bibr" rid="B7">Brown et al., 2020</xref>) and temporal health data (<xref ref-type="bibr" rid="B37">Rasmy et al., 2021</xref>). Here, we investigated the effect of applying ANAA on model performance for HF prediction with reduced data sample sizes. We decreased the fine-tuning sample size to 50%, 20%, and 10%, respectively. The performance of the pre-trained Transformer with and without ANAA, was compared on both the MDC and MIMIC-IV datasets. <xref ref-type="fig" rid="F3">Figures 3a</xref>, <xref ref-type="fig" rid="F4">4</xref> shows that ANAA improves the model performance by around 3% in HF prediction on the MIMIC-IV dataset across all data sample sizes. Similarly, <xref ref-type="fig" rid="F3">Figures 3b</xref>, <xref ref-type="fig" rid="F4">4</xref> demonstrates that ANAA consistently outperforms the baseline in HF prediction on the MDC dataset, even with a 50% reduction in training samples. However, its superiority diminishes with less data.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Impact of ANAA on AUC for HF prediction across different fine-tuning sample sizes in the MIMIC-IV and MDC datasets. The red line shows the AUC of a Transformer model pre-trained on MLM without augmentation; the blue line shows the AUC of the same model augmented with ANAA. In MIMIC-IV, MLM&#x0002B;ANAA consistently outperforms the MLM baseline at all sample sizes. In MDC, MLM&#x0002B;ANAA outperforms the baseline up to the 50% training size; at smaller sizes, its performance converges to that of the baseline due to the limited number of HF-positive samples in the MDC dataset. <bold>(a)</bold> AUC values for HF prediction across fine-tuning sample sizes on the MIMIC-IV test set. <bold>(b)</bold> AUC values for HF prediction across fine-tuning sample sizes on the MDC test set.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1663484-g0003.tif">
<alt-text>Two line graphs compare model AUC performance. The left graph shows two models, MLM (red) and MLM&#x0002B;ANAA (blue), across training sample sizes from 10% to 100%. MLM&#x0002B;ANAA consistently outperforms MLM. The right graph follows a similar pattern, with both models increasing performance as sample size increases, and MLM&#x0002B;ANAA showing higher values. Shaded areas indicate confidence intervals.</alt-text>
</graphic>
</fig>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Comparison of the impact of ANAA on self-attention score distributions in fine-tuned models. Attention scores from each head are individually scaled to the [0, 1] range before plotting their distributions. <bold>(a)</bold> Pre-trained Transformer &#x0002B; Smoothed noise. <bold>(b)</bold> Pre-trained Transformer &#x0002B; ANAA. <bold>(c)</bold> Pre-trained Transformer &#x0002B; ANAA. <bold>(d)</bold> Pre-trained Transformer &#x0002B; ANAA. <bold>(e)</bold> Pre-trained Transformer &#x0002B; RNA. <bold>(f)</bold> Pre-trained Transformer &#x0002B; RNA. <bold>(g)</bold> Pre-trained Transformer &#x0002B; RNA. <bold>(h)</bold> Pre-trained Transformer &#x0002B; RNA. <bold>(i)</bold> Pre-trained Transformer. <bold>(j)</bold> Pre-trained Transformer. <bold>(k)</bold> Pre-trained Transformer. <bold>(l)</bold> Pre-trained Transformer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1663484-g0004.tif">
<alt-text>Twelve histogram plots showing distributions from various prediction models using transformers. Panels (a) to (d) compare different methods, including Smoothed noise and ANAA variations. Panels (e) to (h) use RNA methods. Panels (i) to (l) include HF, AD, and PLS predictions, with data from both MDC and MIMIC-IV datasets. Each plot varies in shape, illustrating different data distributions.</alt-text>
</graphic>
</fig>
</sec>
<sec>
<title>5.6 VS hidden representation augmentation</title>
<p>We first compared ANAA with other hidden representation augmentation methods proposed for augmenting different layers of pre-trained Transformers. Specifically, we assess the impact of injecting noise into various components of the network, such as hidden layers and feedforward modules, as explored in works like HyPe (<xref ref-type="bibr" rid="B47">Yuan et al., 2022</xref>) and Neftune (<xref ref-type="bibr" rid="B21">Jain et al., 2023</xref>). Our objective is to evaluate whether augmenting self-attention scores, where contextual dependencies are explicitly encoded, is more effective than augmenting other internal representations.</p>
<p>As shown in <xref ref-type="table" rid="T3">Table 3</xref>, although NefTune (<xref ref-type="bibr" rid="B21">Jain et al., 2023</xref>) enhances the performance of pre-trained Transformers in HF prediction across both datasets, ANAA consistently outperforms both NefTune and feedforward noise augmentation in predicting outcomes. While ANAA demonstrates superior performance in this context, NefTune has the advantage of being computationally lighter. However, since both methods are applied during fine-tuning, the computational demands are not a significant concern.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Comparing ANAA with naive masking and other hidden representation augmentation methods.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Model/downstream task</bold></th>
<th valign="top" align="center"><bold>HF prediction (MDC)</bold></th>
<th valign="top" align="center"><bold>HF prediction (MIMIC-IV)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Transformer pre-trained on MLM</td>
<td valign="top" align="center">72.2 (2.5)</td>
<td valign="top" align="center">85.2 (1.1)</td>
</tr> <tr>
<td valign="top" align="left">Transformer pre-trained on MLM&#x0002B; Naive masking</td>
<td valign="top" align="center">70.00 (1.5)</td>
<td valign="top" align="center">85.1 (0.7)</td>
</tr> <tr>
<td valign="top" align="left">Transformer pre-trained on MLM&#x0002B; DropAttention</td>
<td valign="top" align="center">69.7 (1.1)</td>
<td valign="top" align="center">84.9 (1.3)</td>
</tr> <tr>
<td valign="top" align="left">Transformer pre-trained on MLM&#x0002B; NEFTune (&#x003B1; &#x0003D; 5)</td>
<td valign="top" align="center">73.6 (3.2)</td>
<td valign="top" align="center">85.2 (0.7)</td>
</tr> <tr>
<td valign="top" align="left">Transformer pre-trained on MLM&#x0002B; NEFTune (&#x003B1; &#x0003D; 10)</td>
<td valign="top" align="center">73.1 (1.7)</td>
<td valign="top" align="center">85.5 (0.4)</td>
</tr> <tr>
<td valign="top" align="left">Transformer pre-trained on MLM&#x0002B; noise in the feedforward (&#x003B1; &#x0003D; 5)</td>
<td valign="top" align="center">73.7 (2.2)</td>
<td valign="top" align="center">85.0 (1.2)</td>
</tr> <tr>
<td valign="top" align="left">Transformer pre-trained on MLM&#x0002B; noise in the feedforward (&#x003B1; &#x0003D; 10)</td>
<td valign="top" align="center">72.5 (4.4)</td>
<td valign="top" align="center">84.5 (0.8)</td>
</tr> <tr>
<td valign="top" align="left">Transformer pre-trained on MLM&#x0002B; ANAA</td>
<td valign="top" align="center"><bold>74.5 (2.9)</bold></td>
<td valign="top" align="center"><bold>87.2 (0.4)</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The table shows the average AUC values (%) and standard deviation across HF prediction tasks on the MDC and MIMIC-IV datasets. Boldface indicates the best-performing model.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>5.7 VS naive masking</title>
<p>Randomly masking the attention score matrix during training can be seen as an extreme form of RNA augmentation. Instead of adding normal noise to perturb relationships between events in a sequence, naive masking directly disrupts these relationships by summing each element with 0 or &#x02212;<italic>A</italic><sub><italic>h</italic><sub><italic>i, j</italic></sub></sub>, effectively breaking the connections between tokens. We compared our method with naive self-attention masking, as described by (<xref ref-type="bibr" rid="B45">Wu et al. 2023</xref>), which introduces a bias in the structure of self-attentions:</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="center"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">softmax</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mi>M</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mo>-</mml:mo><mml:mi>&#x0221E;</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>M</italic><sub><italic>i, j</italic></sub> &#x0003D; &#x02212;&#x0221E; with <italic>p</italic> &#x0003D; 0.2, optimized based on performance on the fine-tuning dataset. We extended it to DropAttention (<xref ref-type="bibr" rid="B49">Zehui et al., 2019</xref>), which expands the mask with a span length &#x003C9; and we set &#x003C9; &#x0003D; Kernel size. However, neither naive masking nor DropAttention improved the performance of the pre-trained Transformer for HF prediction on the MDC and MIMIC-IV datasets. Instead, these methods only increased the number of training iterations required for convergence (see <xref ref-type="table" rid="T3">Table 3</xref>). While these techniques can help mitigate overfitting, their overly aggressive regularization often disrupts critical dependencies within sequences, leading to unstable training and poorer overall performance, especially on complex healthcare prediction tasks. In contrast, ANAA introduces controlled perturbations that balance the attention distribution and prevent over-reliance on specific patterns, thereby preserving essential relationships in the data and promoting more robust and effective representations (see <xref ref-type="supplementary-material" rid="SM1">Supplementary material Section 1.7</xref> for a justification of ANAA as a structured variant of dropout).</p>
</sec>
<sec>
<title>5.8 Effect of ANAA on self-attention behavior</title>
<p>Analyzing self-attention weights and attention score matrices can highlight how Transformers prioritize relationships between events, shedding light on their internal logic and behavior (<xref ref-type="bibr" rid="B12">Clark et al., 2019</xref>; <xref ref-type="bibr" rid="B26">Kovaleva et al., 2019</xref>; <xref ref-type="bibr" rid="B18">Hao et al., 2021</xref>). To assess the effect of ANAA and compare it with normal noise injection (RNA), we analyzed attention score distributions in models fine-tuned on all downstream tasks.</p>
<p>We plotted histograms of attention scores across all heads and samples from the test split, scaling each head&#x00027;s scores to the [0, 1] range (<xref ref-type="fig" rid="F4">Figure 4</xref>). In the bottom row of the figure, we observe that attention scores from the fine-tuned vanilla Transformer tend to cluster near 0 or 1, forming a near-binary (binomial-like) distribution. This pattern suggests overconfidence and limited exploration of dependencies across tokens.</p>
<p>In contrast, the middle row shows that RNA&#x02014;injecting Gaussian noise during training&#x02014;broadens the distribution, encouraging attention heads to explore more diverse and weaker connections. This leads to overlapping attention patterns and increased representation diversity. A mathematical explanation for this phenomenon is provided in <xref ref-type="supplementary-material" rid="SM1">Supplementary material Section 1.5</xref>.</p>
<p>The top row demonstrates the effect of ANAA, which combines noise injection with Gaussian smoothing. This operation retains the diversity introduced by noise while stabilizing the attention pattern, restoring smoother and more informative distributions. The smoothing step dampens extreme noise while allowing the model to refine its exploration of differnt interactions.</p>
<p>To further investigate, we visualized the attention score matrices from models fine-tuned on a representative test sample from the HF prediction task on the MDC dataset (<xref ref-type="fig" rid="F5">Figure 5</xref>). Comparing the original and smoothed attention scores, we observe that ANAA promotes broader attention coverage, with activation scores scaled to the [0, 1] range. <xref ref-type="fig" rid="F5">Figure 5</xref> illustrates an attention head from the first layer, confirming that ANAA leads to more distributed attention patterns. Additional examples from the MIMIC-IV dataset are provided in the <xref ref-type="supplementary-material" rid="SM1">Supplementary material Section 1.10</xref>.</p>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Comparing the impact of ANAA on the self-attention score weights for five fine-tuned models on HF prediction on the MDC dataset for a specific test sample. Here, the attention scores are scaled within 0 and 1. <bold>(a)</bold> Transformer. <bold>(b)</bold> Transformer &#x0002B; ANAA. <bold>(c)</bold> Pre-trained Transformer. <bold>(d)</bold> Pre-trained Transformer &#x0002B; RNA. <bold>(e)</bold> Pre-trained Transformer &#x0002B; ANAA.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1663484-g0005.tif">
<alt-text>Five heatmap panels comparing different transformer models. (a) Transformer shows grid-like patterns. (b) Transformer&#x0002B;ANAA has a more dispersed pattern. (c) Pre-trained Transformer shows distinct horizontal lines. (d) Pre-trained Transformer&#x0002B;RNA appears densely patterned. (e) Pre-trained Transformer&#x0002B;ANAA exhibits a speckled texture. Color scales range from purple to orange.</alt-text>
</graphic>
</fig>
<p>However, it is important to note that the heat-maps in <xref ref-type="fig" rid="F4">Figures 4</xref>, <xref ref-type="fig" rid="F5">5</xref> are intended as qualitative diagnostics of how ANAA redistributes attention&#x02013;not to explain the model&#x00027;s decisions. As shown in prior work, attention weights can often be manipulated without affecting model outputs, meaning they are not a reliable source of explanation (<xref ref-type="bibr" rid="B18">Hao et al., 2021</xref>; <xref ref-type="bibr" rid="B22">Jain and Wallace, 2019</xref>; <xref ref-type="bibr" rid="B39">Serrano and Smith, 2019</xref>). We therefore interpret these visualizations only as evidence that ANAA breaks the near-binary pattern observed in the baseline model; attributing clinical relevance to specific codes and specific codes with each other in this context would require dedicated methods such as Integrated Gradients (<xref ref-type="bibr" rid="B41">Sundararajan et al., 2017</xref>) and can be investigated further in future work.</p>
<sec>
<title>5.8.1 Effect of ANAA on the receptive field</title>
<p>The self-attention mechanism is designed to capture both long and short-range dependencies effectively. To quantitatively assess the impact of RNA and ANAA on the receptive field, we plot the median values of attention score matrix <italic>A</italic><sub><italic>h</italic></sub> for each event with respect to all previous and subsequent events (<italic>i</italic>&#x02212;<italic>j, A</italic><sub><italic>h</italic><sub><italic>i, j</italic></sub></sub>) -<italic>i, j</italic> are positions of <italic>e</italic><sub><italic>i</italic></sub>, <italic>e</italic><sub><italic>j</italic></sub> in the sequence of events-across all test samples for HF and AD predictions on the MDC (<xref ref-type="fig" rid="F6">Figure 6</xref>). Transformers pre-trained on MLM typically allocate more attention weight to recent events, often in a monotonous fashion. Incorporating RNA regularization reduces the steepness of this attention distribution, allowing events to receive more balanced attention, not solely based on their proximity to recent events. Ultimately, applying ANAA, preserves the benefits of RNA by providing a more equal distribution of attention within a local neighborhood, while simultaneously reducing the emphasis on very distant past events.</p>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>Impact of ANAA on the receptive field of the self-attentions for HF and AD prediction on the MDC dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1663484-g0006.tif">
<alt-text>Two line graphs compare median attention against distance for different models. The left graph shows HF prediction with lines for MLM, MLM &#x0002B; RNA, and MLM &#x0002B; ANAA, which has the highest median attention. The right graph depicts AD prediction, where MLM &#x0002B; ANAA also has the highest median attention. Both graphs use negative to positive distance values on the x-axis. </alt-text>
</graphic>
</fig>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="s6">
<title>6 Discussion</title>
<p>This study demonstrates that ANAA&#x02014;a simple two-step fine-tuning augmentation&#x02014;consistently enhances the discriminative performance of pre-trained Transformers on longitudinal EHR data, without altering their architecture. Compared to the hidden representation augmentation and a range of established regularizers, it yields superior results over vanilla fine-tuning. ANAA produced consistent AUC gains on HF and AD prediction tasks in two different EHR corpora (MDC and MIMIC-IV). On HF prediction, for example, the MLM-pre-trained baseline rose from 72.2 to 74.5 AUC on MDC and from 85.2 to 87.2 AUC on MIMIC-IV after applying ANAA. These gains persisted even under label-scarce conditions, maintaining &#x0007E;3 percentage-point improvements. These findings suggest that judicious noise injection at the level of self-attention&#x02014;followed by controlled Gaussian smoothing&#x02014;can encourage pre-trained transformers to explore and learn more robust, generalizable patterns.</p>
<p>We further investigated that Transformers pre-trained via MLM&#x02014;while typically outperforming models without pre-training&#x02014;can exhibit overconfident, sparse attention patterns during fine-tuning. Attention histograms reveal that conventional fine-tuning drives many heads toward almost binary (0/1) weights, indicating over-confident, brittle dependencies. ANAA counteracts this by injecting adaptive Gaussian noise, which broadens the attention distribution and encourages heads to sample a richer set of relational cues. The subsequent smoothing step restores coherent structure. As shown analytically in <xref ref-type="supplementary-material" rid="SM1">Supplementary material Section 1.5</xref>, this mechanism effectively acts as a variance-scaled, shifting the attention score distribution from deterministic and binary to probabilistic and continuous, to explore alternative dependencies.</p>
<p>Compared to other augmentation methods such as NEFTune (<xref ref-type="bibr" rid="B21">Jain et al., 2023</xref>) and HyPe (<xref ref-type="bibr" rid="B47">Yuan et al., 2022</xref>)&#x02014;which add noise in the embedding or feed-forward layers&#x02014;ANAA achieves larger and more consistent performance gains. In contrast, naive attention masking or DropAttention (<xref ref-type="bibr" rid="B45">Wu et al., 2023</xref>; <xref ref-type="bibr" rid="B49">Zehui et al., 2019</xref>) degraded results. This highlights the importance of <italic>where</italic> noise is injected: perturbing the self-attention scores&#x02014;the core mechanism for modeling token interactions&#x02014;yields greater benefit than altering downstream representations.</p>
<p>While ANAA consistently improves performance across the two studied EHR datasets, several caveats remain. First, all experiments were conducted on structured, diagnosis- and medication-coded timelines (MIMIC-IV and MDC); Although our experiments focus on a standard Transformer encoder for clarity and control, ANAA is modular by design and can be integrated into other clinical Transformer models such as BEHRT (<xref ref-type="bibr" rid="B31">Li et al., 2020</xref>), Med-BERT (<xref ref-type="bibr" rid="B37">Rasmy et al., 2021</xref>), or Hi-BEHRT (<xref ref-type="bibr" rid="B30">Li et al., 2022</xref>); exploring such integrations is a promising direction for future work. More broadly, how well ANAA generalizes to other data modalities &#x02013;such as free text, imaging, or genomics &#x02013;and to models pre-trained with alternative objectives such as contrastive learning (e.g., BYOL; <xref ref-type="bibr" rid="B16">Grill et al., 2020</xref>) also remains to be explored. Second, ANAA introduces additional hyperparameters. Although the sensitivity analysis in <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S3</xref> suggests the method is robust across a range of values, some tuning is still required. Third, the computational overhead introduced by noise injection and smoothing increases both memory usage and training time, which may become a limitation for very long sequences or resource-constrained environments. Fourth, in settings with extremely low data regimes or highly unbalanced labels, ANAA&#x00027;s implicit Augmentation provides some benefit but is not sufficient on its own. Finally, the effect of model augmentations, like ANAA, on model interpretability warrants further study, particularly in safety-critical applications.</p></sec>
<sec sec-type="conclusions" id="s7">
<title>7 Conclusion</title>
<p>We introduced <italic>Adaptive Noise-Augmented Attention</italic> (ANAA), a lightweight and effective method for enhancing the fine-tuning of pre-trained Transformers. ANAA directly augments the self-attention scores with adaptive Gaussian noise and applies a smoothing convolution using a Gaussian kernel, encouraging the model to explore more diverse attention patterns while preserving critical dependencies.</p>
<p>We demonstrated that pre-trained Transformers, when fine-tuned on limited EHR datasets, often converge to overly sharp attention distributions&#x02014;overfitting to local patterns and failing to capture broader contextual relationships. ANAA mitigates this by encouraging more diverse and stable attention distributions, leading to better generalization across tasks and data regimes. Extensive experiments on multiple clinical prediction tasks showed that ANAA consistently outperforms conventional regularization and hidden augmentation techniques.</p>
<p>ANAA offers a plug-and-play augmentation mechanism that operates entirely within the attention computation, requiring no modification to the model architecture or computational graph. This makes it particularly suitable for integration with existing pre-trained models.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s8">
<title>Data availability statement</title>
<p>The MIMIC-IV dataset is publicly available from the PhysioNet repository [<ext-link ext-link-type="uri" xlink:href="https://physionet.org/content/mimiciv/2.2/">https://physionet.org/content/mimiciv/2.2/</ext-link>]. The Malmo Diet and Cancer Cohort data that support the findings of this study are not publicly available due to data access restrictions imposed by the Malmo Population-Based Cohorts Joint Database. However, the data are available from the corresponding author upon reasonable request and with permission from the Malmo Population-Based Cohorts Joint Database [<ext-link ext-link-type="uri" xlink:href="https://www.malmo-kohorter.lu.se/malmo-cohorts">https://www.malmo-kohorter.lu.se/malmo-cohorts</ext-link>].</p>
</sec>
<sec sec-type="ethics-statement" id="s9">
<title>Ethics statement</title>
<p>The use of the MDC dataset for this study was approved by the Ethics Review Board of Sweden (Dnr 2023-00503-01). Regarding the MIMIC-IV dataset, all protected health information (PHI) is officially deidentified. It means that the deletion of PHI from structured data sources (e.g., database fields that provide age, genotypic information, and past and current diagnosis and treatment categories) is performed in compliance with the HIPAA (Health Insurance Portability and Accountability Act) standards in order to facilitate public access to the datasets.</p>
</sec>
<sec sec-type="author-contributions" id="s10">
<title>Author contributions</title>
<p>AA: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. FE: Conceptualization, Formal analysis, Funding acquisition, Investigation, Methodology, Resources, Supervision, Validation, Visualization, Writing &#x02013; review &#x00026; editing. MO: Conceptualization, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Supervision, Validation, Visualization, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s11">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This study was conducted as part of the AIR Lund (Artificially Intelligent use of Registers at Lund University) research environment and was funded by the Swedish Research Council (VR, grant 2019-00198). Additional support was provided by CAISR Health, funded by the Knowledge Foundation (KK-stiftelsen) in Sweden (grant 20200208 01 H).</p>
</sec>
<ack><p>We thank Jonas Bj&#x000F6;rk and Olle Melander for facilitating access to the data and for their valuable guidance in understanding and interpreting the dataset.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s12">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s13">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s14">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2025.1663484/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frai.2025.1663484/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Amirahmadi</surname> <given-names>A.</given-names></name> <name><surname>Etminani</surname> <given-names>F.</given-names></name> <name><surname>Bj&#x000F6;rk</surname> <given-names>J.</given-names></name> <name><surname>Melander</surname> <given-names>O.</given-names></name> <name><surname>Ohlsson</surname> <given-names>M.</given-names></name></person-group> (<year>2025</year>). <article-title>Trajectory-ordered objectives for self-supervised representation learning of temporal healthcare data using transformers: Model development and evaluation study</article-title>. <source>JMIR Med. Inform</source>. <volume>13</volume>:<fpage>e68138</fpage>. <pub-id pub-id-type="doi">10.2196/68138</pub-id><pub-id pub-id-type="pmid">40465350</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Amirahmadi</surname> <given-names>A.</given-names></name> <name><surname>Ohlsson</surname> <given-names>M.</given-names></name> <name><surname>Etminani</surname> <given-names>K.</given-names></name></person-group> (<year>2023</year>). <article-title>Deep learning prediction models based on ehr trajectories: a systematic review</article-title>. <source>J. Biomed. Inform</source>. <volume>144</volume>:<fpage>104430</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2023.104430</pub-id><pub-id pub-id-type="pmid">37380061</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Amos</surname> <given-names>I.</given-names></name> <name><surname>Berant</surname> <given-names>J.</given-names></name> <name><surname>Gupta</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Never train from scratch: fair comparison of long-sequence models requires data-driven priors</article-title>. <source>arXiv preprint arXiv:2310.02980</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2310.02980</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berglund</surname> <given-names>G.</given-names></name> <name><surname>Elmst&#x000E5;hl</surname> <given-names>S.</given-names></name> <name><surname>Janzon</surname> <given-names>L.</given-names></name> <name><surname>Larsson</surname> <given-names>S.</given-names></name></person-group> (<year>1993</year>). <article-title>The malmo diet and cancer study. Design and feasibility</article-title>. <source>J. Intern. Med</source>. <volume>233</volume>, <fpage>45</fpage>&#x02013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-2796.1993.tb00647.x</pub-id><pub-id pub-id-type="pmid">8429286</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boll</surname> <given-names>H. O.</given-names></name> <name><surname>Amirahmadi</surname> <given-names>A.</given-names></name> <name><surname>Ghazani</surname> <given-names>M. M.</given-names></name> <name><surname>de Morais</surname> <given-names>W. O.</given-names></name> <name><surname>de Freitas</surname> <given-names>E. P.</given-names></name> <name><surname>Soliman</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Graph neural networks for clinical risk prediction based on electronic health records: a survey</article-title>. <source>J. Biomed. Inform</source>. <volume>151</volume>:<fpage>104616</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2024.104616</pub-id><pub-id pub-id-type="pmid">38423267</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bommasani</surname> <given-names>R.</given-names></name> <name><surname>Hudson</surname> <given-names>D. A.</given-names></name> <name><surname>Adeli</surname> <given-names>E.</given-names></name> <name><surname>Altman</surname> <given-names>R.</given-names></name> <name><surname>Arora</surname> <given-names>S.</given-names></name> <name><surname>von Arx</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>On the opportunities and risks of foundation models</article-title>. <source>arXiv preprint arXiv:2108.07258</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2108.07258</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname> <given-names>T.</given-names></name> <name><surname>Mann</surname> <given-names>B.</given-names></name> <name><surname>Ryder</surname> <given-names>N.</given-names></name> <name><surname>Subbiah</surname> <given-names>M.</given-names></name> <name><surname>Kaplan</surname> <given-names>J. D.</given-names></name> <name><surname>Dhariwal</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Language models are few-shot learners</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>33</volume>, <fpage>1877</fpage>&#x02013;<lpage>1901</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2005.14165</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Camuto</surname> <given-names>A.</given-names></name> <name><surname>Willetts</surname> <given-names>M.</given-names></name> <name><surname>Simsekli</surname> <given-names>U.</given-names></name> <name><surname>Roberts</surname> <given-names>S. J.</given-names></name> <name><surname>Holmes</surname> <given-names>C. C.</given-names></name></person-group> (<year>2020</year>). <article-title>Explicit regularisation in gaussian noise injections</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>33</volume>, <fpage>16603</fpage>&#x02013;<lpage>16614</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2007.07368</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>Z.</given-names></name> <name><surname>Yang</surname> <given-names>D.</given-names></name></person-group> (<year>2020</year>). <article-title>Mixtext: linguistically-informed interpolation of hidden space for semi-supervised text classification</article-title>. <source>arXiv preprint arXiv:2004.12239</source>. <pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.194</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Xie</surname> <given-names>S.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;An empirical study of training self-supervised vision transformers,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source> (<publisher-loc>Los Alamitos, CA</publisher-loc>: <publisher-name>IEEE Computer Society (Conference Publishing Services</publisher-name>)), <fpage>9640</fpage>&#x02013;<lpage>9649</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00950</pub-id><pub-id pub-id-type="pmid">36383492</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Choi</surname> <given-names>E.</given-names></name> <name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Dusenberry</surname> <given-names>M.</given-names></name> <name><surname>Flores</surname> <given-names>G.</given-names></name> <name><surname>Xue</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Learning the graphical structure of electronic health records with graph convolutional transformer</article-title>. <source>Proc. AAAI Conf. Artif. Intell</source>. <volume>34</volume>, <fpage>606</fpage>&#x02013;<lpage>613</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v34i01.5400</pub-id><pub-id pub-id-type="pmid">39359569</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clark</surname> <given-names>K.</given-names></name> <name><surname>Khandelwal</surname> <given-names>U.</given-names></name> <name><surname>Levy</surname> <given-names>O.</given-names></name> <name><surname>Manning</surname> <given-names>C. D.</given-names></name></person-group> (<year>2019</year>). <article-title>What does bert look at? an analysis of bert&#x00027;s attention</article-title>. <source>arXiv preprint arXiv:1906.04341</source>. <pub-id pub-id-type="doi">10.18653/v1/W19-4828</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Devlin</surname> <given-names>J.</given-names></name> <name><surname>Chang</surname> <given-names>M.-W.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Toutanova</surname> <given-names>K.</given-names></name></person-group> (<year>2018</year>). <article-title>Bert: Pre-training of deep bidirectional transformers for language understanding</article-title>. <source>arXiv preprint arXiv:1810.04805</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1810.04805</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>J.</given-names></name> <name><surname>Ma</surname> <given-names>S.</given-names></name> <name><surname>Dong</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Longnet: scaling transformers to 1,000,000,000 tokens</article-title>. <source>arXiv preprint arXiv:2307.02486</source>. <pub-id pub-id-type="doi">10.14218/JCTH.2022.00006S</pub-id><pub-id pub-id-type="pmid">37577238</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname> <given-names>A.</given-names></name> <name><surname>Beyer</surname> <given-names>L.</given-names></name> <name><surname>Kolesnikov</surname> <given-names>A.</given-names></name> <name><surname>Weissenborn</surname> <given-names>D.</given-names></name> <name><surname>Zhai</surname> <given-names>X.</given-names></name> <name><surname>Unterthiner</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>An image is worth 16 &#x000D7; 16 words: transformers for image recognition at scale</article-title>. <source>arXiv preprint arXiv</source>:2010.11929. <pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grill</surname> <given-names>J.-B.</given-names></name> <name><surname>Strub</surname> <given-names>F.</given-names></name> <name><surname>Altch&#x000E9;</surname> <given-names>F.</given-names></name> <name><surname>Tallec</surname> <given-names>C.</given-names></name> <name><surname>Richemond</surname> <given-names>P.</given-names></name> <name><surname>Buchatskaya</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Bootstrap your own latent-a new approach to self-supervised learning</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>33</volume>, <fpage>21271</fpage>&#x02013;<lpage>21284</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2006.07733</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>L. L.</given-names></name> <name><surname>Steinberg</surname> <given-names>E.</given-names></name> <name><surname>Fleming</surname> <given-names>S. L.</given-names></name> <name><surname>Posada</surname> <given-names>J.</given-names></name> <name><surname>Lemmon</surname> <given-names>J.</given-names></name> <name><surname>Pfohl</surname> <given-names>S. R.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Ehr foundation models improve robustness in the presence of temporal distribution shift</article-title>. <source>Sci. Rep</source>. <volume>13</volume>:<fpage>3767</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-30820-8</pub-id><pub-id pub-id-type="pmid">36882576</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hao</surname> <given-names>Y.</given-names></name> <name><surname>Dong</surname> <given-names>L.</given-names></name> <name><surname>Wei</surname> <given-names>F.</given-names></name> <name><surname>Xu</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <article-title>Self-attention attribution: interpreting information interactions inside transformer</article-title>. <source>Proc. AAAI Conf. Artif. Intell</source>. <volume>35</volume>, <fpage>12963</fpage>&#x02013;<lpage>12971</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v35i14.17533</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hassani</surname> <given-names>A.</given-names></name> <name><surname>Walton</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Shi</surname> <given-names>H.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Neighborhood attention transformer,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Los Alamitos, CA</publisher-loc>: <publisher-name>IEEE Computer Society (Conference Publishing Services</publisher-name>)), <fpage>6185</fpage>&#x02013;<lpage>6194</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52729.2023.00599</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hassani</surname> <given-names>A.</given-names></name> <name><surname>Walton</surname> <given-names>S.</given-names></name> <name><surname>Shah</surname> <given-names>N.</given-names></name> <name><surname>Abuduweili</surname> <given-names>A.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Shi</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>Escaping the big data paradigm with compact transformers</article-title>. <source>arXiv preprint arXiv:2104.05704</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2104.05704</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jain</surname> <given-names>N.</given-names></name> <name><surname>Chiang</surname> <given-names>P.-y.</given-names></name> <name><surname>Wen</surname> <given-names>Y.</given-names></name> <name><surname>Kirchenbauer</surname> <given-names>J.</given-names></name> <name><surname>Chu</surname> <given-names>H.-M.</given-names></name> <name><surname>Somepalli</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Neftune: Noisy embeddings improve instruction finetuning</article-title>. <source>arXiv preprint arXiv:2310.05914</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2310.05914</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jain</surname> <given-names>S.</given-names></name> <name><surname>Wallace</surname> <given-names>B. C.</given-names></name></person-group> (<year>2019</year>). <article-title>Attention is not explanation</article-title>. <source>arXiv preprint arXiv:1902.10186</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1902.10186</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Johnson</surname> <given-names>A.</given-names></name> <name><surname>Bulgarelli</surname> <given-names>L.</given-names></name> <name><surname>Pollard</surname> <given-names>T.</given-names></name> <name><surname>Horng</surname> <given-names>S.</given-names></name> <name><surname>Celi</surname> <given-names>L. A.</given-names></name> <name><surname>Mark</surname> <given-names>R.</given-names></name></person-group> (<year>2020</year>). <source>Mimic-iv. PhysioNet</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://physionet.org/content/mimiciv/1.0/">https://physionet.org/content/mimiciv/1.0/</ext-link> (Accessed August 23, 2021).</citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>K. G.</given-names></name> <name><surname>Lee</surname> <given-names>B. T.</given-names></name></person-group> (<year>2024</year>). <article-title>Self-attention with temporal prior: can we learn more from the arrow of time?</article-title> <source>Front. Artif. Intell</source>. <volume>7</volume>:<fpage>1397298</fpage>. <pub-id pub-id-type="doi">10.3389/frai.2024.1397298</pub-id><pub-id pub-id-type="pmid">39165902</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kong</surname> <given-names>K.</given-names></name> <name><surname>Li</surname> <given-names>G.</given-names></name> <name><surname>Ding</surname> <given-names>M.</given-names></name> <name><surname>Wu</surname> <given-names>Z.</given-names></name> <name><surname>Zhu</surname> <given-names>C.</given-names></name> <name><surname>Ghanem</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;Robust optimization as data augmentation for large-scale graphs,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Los Alamitos, CA</publisher-loc>: <publisher-name>IEEE Computer Society (Conference Publishing Services</publisher-name>)), <fpage>60</fpage>&#x02013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.00016</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kovaleva</surname> <given-names>O.</given-names></name> <name><surname>Romanov</surname> <given-names>A.</given-names></name> <name><surname>Rogers</surname> <given-names>A.</given-names></name> <name><surname>Rumshisky</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>Revealing the dark secrets of bert</article-title>. <source>arXiv preprint arXiv</source>:1908.08593. <pub-id pub-id-type="doi">10.18653/v1/D19-1445</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lan</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>M.</given-names></name> <name><surname>Goodman</surname> <given-names>S.</given-names></name> <name><surname>Gimpel</surname> <given-names>K.</given-names></name> <name><surname>Sharma</surname> <given-names>P.</given-names></name> <name><surname>Soricut</surname> <given-names>R.</given-names></name></person-group> (<year>2019</year>). <article-title>Albert: a lite bert for self-supervised learning of language representations</article-title>. <source>arXiv preprint arXiv:1909.11942</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1909.11942</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lewis</surname> <given-names>M.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Goyal</surname> <given-names>N.</given-names></name> <name><surname>Ghazvininejad</surname> <given-names>M.</given-names></name> <name><surname>Mohamed</surname> <given-names>A.</given-names></name> <name><surname>Levy</surname> <given-names>O.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Bart: denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension</article-title>. <source>arXiv preprint arXiv:1910.13461</source>. <pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.703</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Zhou</surname> <given-names>J.</given-names></name> <name><surname>Gao</surname> <given-names>Z.</given-names></name> <name><surname>Hua</surname> <given-names>W.</given-names></name> <name><surname>Fan</surname> <given-names>L.</given-names></name> <name><surname>Yu</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>A scoping review of using large language models (LLMS) to investigate electronic health records (EHRS)</article-title>. <source>arXiv preprint arXiv</source>:2405.03066. <pub-id pub-id-type="doi">10.48550/arXiv.2405.03066</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Mamouei</surname> <given-names>M.</given-names></name> <name><surname>Salimi-Khorshidi</surname> <given-names>G.</given-names></name> <name><surname>Rao</surname> <given-names>S.</given-names></name> <name><surname>Hassaine</surname> <given-names>A.</given-names></name> <name><surname>Canoy</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Hi-behrt: hierarchical transformer-based model for accurate prediction of clinical events using multimodal longitudinal electronic health records</article-title>. <source>IEEE J. Biomed. Health Inform</source>. <volume>27</volume>, <fpage>1106</fpage>&#x02013;<lpage>1117</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2022.3224727</pub-id><pub-id pub-id-type="pmid">36427286</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Rao</surname> <given-names>S.</given-names></name> <name><surname>Solares</surname> <given-names>J. R. A.</given-names></name> <name><surname>Hassaine</surname> <given-names>A.</given-names></name> <name><surname>Ramakrishnan</surname> <given-names>R.</given-names></name> <name><surname>Canoy</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Behrt: transformer for electronic health records</article-title>. <source>Sci. Rep</source>. <volume>10</volume>:<fpage>7155</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-62922-y</pub-id><pub-id pub-id-type="pmid">32346050</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>P.</given-names></name> <name><surname>Yuan</surname> <given-names>W.</given-names></name> <name><surname>Fu</surname> <given-names>J.</given-names></name> <name><surname>Jiang</surname> <given-names>Z.</given-names></name> <name><surname>Hayashi</surname> <given-names>H.</given-names></name> <name><surname>Neubig</surname> <given-names>G.</given-names></name></person-group> (<year>2023</year>). <article-title>Pre-train, prompt, and predict: a systematic survey of prompting methods in natural language processing</article-title>. <source>ACM Comput. Surv</source>. <volume>55</volume>, <fpage>1</fpage>&#x02013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1145/3560815</pub-id></citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Ott</surname> <given-names>M.</given-names></name> <name><surname>Goyal</surname> <given-names>N.</given-names></name> <name><surname>Du</surname> <given-names>J.</given-names></name> <name><surname>Joshi</surname> <given-names>M.</given-names></name> <name><surname>Chen</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Roberta: a robustly optimized bert pretraining approach</article-title>. <source>arXiv preprint arXiv:1907.11692</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1907.11692</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Pang</surname> <given-names>C.</given-names></name> <name><surname>Jiang</surname> <given-names>X.</given-names></name> <name><surname>Kalluri</surname> <given-names>K. S.</given-names></name> <name><surname>Spotnitz</surname> <given-names>M.</given-names></name> <name><surname>Chen</surname> <given-names>R.</given-names></name> <name><surname>Perotte</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Cehr-bert: incorporating temporal information from structured ehr data to improve prediction tasks,&#x0201D;</article-title> in <source>Machine Learning for Health</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>239</fpage>&#x02013;<lpage>260</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><collab>Press O. Smith N. A. Lewis M.</collab></person-group> (<year>2021</year>). <article-title>Train short, test long: attention with linear biases enables input length extrapolation</article-title>. <source>arXiv preprint arXiv:2108.12409</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2108.12409</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Radford</surname> <given-names>A.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Child</surname> <given-names>R.</given-names></name> <name><surname>Luan</surname> <given-names>D.</given-names></name> <name><surname>Amodei</surname> <given-names>D.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name></person-group> (<year>2019</year>). <source>Language Models Are Unsupervised Multitask Learners</source>. Technical Report. OpenAI. Available online at: <ext-link ext-link-type="uri" xlink:href="https://openai.com/index/better-language-models/">https://openai.com/index/better-language-models/</ext-link> (Accessed September 04, 2025).<pub-id pub-id-type="pmid">35637722</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rasmy</surname> <given-names>L.</given-names></name> <name><surname>Xiang</surname> <given-names>Y.</given-names></name> <name><surname>Xie</surname> <given-names>Z.</given-names></name> <name><surname>Tao</surname> <given-names>C.</given-names></name> <name><surname>Zhi</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>Med-bert: pretrained contextualized embeddings on large-scale structured electronic health records for disease prediction</article-title>. <source>NPJ Digit. Med</source>. <volume>4</volume>:<fpage>86</fpage>. <pub-id pub-id-type="doi">10.1038/s41746-021-00455-y</pub-id><pub-id pub-id-type="pmid">34017034</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ren</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Zhao</surname> <given-names>W. X.</given-names></name> <name><surname>Wu</surname> <given-names>N.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Rapt: pre-training of time-aware transformer for learning robust healthcare representation,&#x0201D;</article-title> in <source>Proceedings of the 27th ACM SIGKDD Conference on Knowledge Discovery and Data Mining</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery (ACM</publisher-name>)), <fpage>3503</fpage>&#x02013;<lpage>3511</lpage>. <pub-id pub-id-type="doi">10.1145/3447548.3467069</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Serrano</surname> <given-names>S.</given-names></name> <name><surname>Smith</surname> <given-names>N. A.</given-names></name></person-group> (<year>2019</year>). Is attention interpretable? <italic>arXiv preprint arXiv:1906.03731</italic>. <pub-id pub-id-type="doi">10.48550/arXiv.1906.03731</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Su</surname> <given-names>J.</given-names></name> <name><surname>Ahmed</surname> <given-names>M.</given-names></name> <name><surname>Lu</surname> <given-names>Y.</given-names></name> <name><surname>Pan</surname> <given-names>S.</given-names></name> <name><surname>Bo</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name></person-group> (<year>2024</year>). <article-title>Roformer: enhanced transformer with rotary position embedding</article-title>. <source>Neurocomputing</source> <volume>568</volume>:<fpage>127063</fpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2023.127063</pub-id></citation>
</ref>
<ref id="B41">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sundararajan</surname> <given-names>M.</given-names></name> <name><surname>Taly</surname> <given-names>A.</given-names></name> <name><surname>Yan</surname> <given-names>Q.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Axiomatic attribution for deep networks,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>Sydney, NSW</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>3319</fpage>&#x02013;<lpage>3328</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Touvron</surname> <given-names>H.</given-names></name> <name><surname>Cord</surname> <given-names>M.</given-names></name> <name><surname>Douze</surname> <given-names>M.</given-names></name> <name><surname>Massa</surname> <given-names>F.</given-names></name> <name><surname>Sablayrolles</surname> <given-names>A.</given-names></name> <name><surname>J&#x000E9;gou</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Training data-efficient image transformers and distillation through attention,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>Vienna</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>10347</fpage>&#x02013;<lpage>10357</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaswani</surname> <given-names>A.</given-names></name> <name><surname>Shazeer</surname> <given-names>N.</given-names></name> <name><surname>Parmar</surname> <given-names>N.</given-names></name> <name><surname>Uszkoreit</surname> <given-names>J.</given-names></name> <name><surname>Jones</surname> <given-names>L.</given-names></name> <name><surname>Gomez</surname> <given-names>A. N.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Attention is all you need</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>30</volume>, <fpage>5999</fpage>&#x02013;<lpage>6009</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1706.03762</pub-id></citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wornow</surname> <given-names>M.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Thapa</surname> <given-names>R.</given-names></name> <name><surname>Patel</surname> <given-names>B.</given-names></name> <name><surname>Steinberg</surname> <given-names>E.</given-names></name> <name><surname>Fleming</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>The shaky foundations of large language models and foundation models for electronic health records</article-title>. <source>NPJ Digit. Med</source>. <volume>6</volume>:<fpage>135</fpage>. <pub-id pub-id-type="doi">10.1038/s41746-023-00879-8</pub-id><pub-id pub-id-type="pmid">37516790</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>H.</given-names></name> <name><surname>Ding</surname> <given-names>R.</given-names></name> <name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Xie</surname> <given-names>P.</given-names></name> <name><surname>Huang</surname> <given-names>F.</given-names></name> <name><surname>Zhang</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Adversarial self-attention for language understanding</article-title>. <source>Proc. AAAI Conf. Artif. Intell</source>. <volume>37</volume>, <fpage>13727</fpage>&#x02013;<lpage>13735</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v37i11.26608</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiao</surname> <given-names>C.</given-names></name> <name><surname>Choi</surname> <given-names>E.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>Opportunities and challenges in developing deep learning models using electronic health records data: a systematic review</article-title>. <source>J. Am. Med. Inform. Assoc</source>. <volume>25</volume>, <fpage>1419</fpage>&#x02013;<lpage>1428</lpage>. <pub-id pub-id-type="doi">10.1093/jamia/ocy068</pub-id><pub-id pub-id-type="pmid">29893864</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yuan</surname> <given-names>H.</given-names></name> <name><surname>Yuan</surname> <given-names>Z.</given-names></name> <name><surname>Tan</surname> <given-names>C.</given-names></name> <name><surname>Huang</surname> <given-names>F.</given-names></name> <name><surname>Huang</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>Hype: better pre-trained language model fine-tuning with hidden representation perturbation</article-title>. <source>arXiv preprint arXiv:2212.08853</source>. <pub-id pub-id-type="doi">10.18653/v1/2023.acl-long.182</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yuanyuan</surname> <given-names>Z.</given-names></name> <name><surname>Adel</surname> <given-names>B.</given-names></name> <name><surname>Mina</surname> <given-names>B.</given-names></name> <name><surname>Jamil</surname> <given-names>Z.</given-names></name> <name><surname>Hugues</surname> <given-names>T.</given-names></name> <name><surname>Lydie</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>A scoping review of self-supervised representation learning for clinical decision making using ehr categorical data</article-title>. <source>NPJ Digit. Med</source>. <volume>8</volume>, <fpage>1</fpage>&#x02013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1038/s41746-025-01692-1</pub-id><pub-id pub-id-type="pmid">40517140</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zehui</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>P.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Qiu</surname> <given-names>X.</given-names></name> <name><surname>Huang</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). <article-title>Dropattention: a regularization method for fully-connected self-attention networks</article-title>. <source>arXiv preprint arXiv</source>:1907.11065. <pub-id pub-id-type="doi">10.48550/arXiv.1907.11065</pub-id></citation>
</ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>C.</given-names></name> <name><surname>Cheng</surname> <given-names>Y.</given-names></name> <name><surname>Gan</surname> <given-names>Z.</given-names></name> <name><surname>Sun</surname> <given-names>S.</given-names></name> <name><surname>Goldstein</surname> <given-names>T.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>Freelb: enhanced adversarial training for natural language understanding</article-title>. <source>arXiv preprint arXiv:1909.11764</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1909.11764</pub-id></citation>
</ref>
<ref id="B51">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>W.</given-names></name> <name><surname>Razavian</surname> <given-names>N.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Variationally regularized graph-based representation learning for electronic health records,&#x0201D;</article-title> in <source>Proceedings of the Conference on Health, Inference, and Learning</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery (ACM)</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1145/3450439.3451855</pub-id></citation>
</ref>
</ref-list>
</back>
</article>