<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Public Health</journal-id>
<journal-title>Frontiers in Public Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Public Health</abbrev-journal-title>
<issn pub-type="epub">2296-2565</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpubh.2024.1520343</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Public Health</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Multimodal approach to public health interventions using EGG and mobile health technologies</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Xiao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Liu</surname> <given-names>Han</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Sun</surname> <given-names>Mingyang</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2883454/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Feng</surname> <given-names>Shuangyi</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Physical Education Institute, Yunnan Minzu University, Kunming</institution>, <addr-line>Yunnan</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>The Catholic University of Korea</institution>, <addr-line>Seoul</addr-line>, <country>Republic of Korea</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Dilbag Singh, New York University, United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Naveen Kumari, Punjabi University, India</p>
<p>Arjun Singh, Manipal University Jaipur, India</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Han Liu <email>lh930&#x00040;126.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>01</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1520343</elocation-id>
<history>
<date date-type="received">
<day>31</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>12</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Zhang, Liu, Sun and Feng.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zhang, Liu, Sun and Feng</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Public health interventions increasingly integrate multimodal data sources, such as Electroencephalogram (EEG) data, to enhance monitoring and predictive capabilities for mental health conditions. However, traditional models often face challenges with the complexity and high dimensionality of EEG signals. While recent advancements like Contrastive Language-lmage Pre-training(CLIP) models excel in cross-modal understanding, their application to EEG-based tasks remains limited due to the unique characteristics of EEG data.</p>
</sec>
<sec>
<title>Methods</title>
<p>In response, we introduce PH-CLIP (Public Health Contrastive Language-lmage Pretraining), a novel framework that combines CLIP&#x00027;s representational power with a multi-scale fusion mechanism designed specifically for EEG data within mobile health technologies. PH-CLIP employs hierarchical feature extraction to capture the temporal dynamics of EEG signals, aligning them with contextually relevant textual descriptions for improved public health insights. Through a multi-scale fusion layer, PH-CLIP enhances interpretability and robustness in EEG embeddings, thereby supporting more accurate and scalable interventions across diverse public health applications.</p>
</sec>
<sec>
<title>Results and discussion</title>
<p>Experimental results indicate that PH-CLIP achieves significant improvements in EEG classification accuracy and mental health prediction efficiency compared to leading EEG analysis models. This framework positions PH-CLIP as a transformative tool in public health monitoring, with the potential to advance large-scale mental health interventions through integrative mobile health technologies.</p>
</sec></abstract>
<kwd-group>
<kwd>public health interventions</kwd>
<kwd>PH-CLIP</kwd>
<kwd>EEG signal analysis</kwd>
<kwd>multi-scale fusion mechanism</kwd>
<kwd>mobile health technologies</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="6"/>
<equation-count count="44"/>
<ref-count count="37"/>
<page-count count="20"/>
<word-count count="13638"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Digital Public Health</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>The need for robust, scalable, and interpretable systems to analyze and interpret electroencephalography (EEG) signals in public health applications has become increasingly critical. EEG provides a non-invasive, real-time monitoring approach, especially useful for mental health and neurological assessments (<xref ref-type="bibr" rid="B1">1</xref>). However, traditional approaches to EEG analysis are challenged by the complex, high-dimensional nature of EEG data, and the requirements for high accuracy and generalizability across diverse populations (<xref ref-type="bibr" rid="B2">2</xref>). PH-CLIP (Public Health Contrastive Language-Image Pretraining) with multi-scale fusion on EEG is proposed to enhance the scalability and precision of EEG interpretation by leveraging recent advances in multi-modal and multi-scale deep learning. This approach not only addresses issues of scalability but also aims to improve cross-population generalization by incorporating scalable contrastive language-image models (CLIP) adapted for EEG data, ultimately offering a powerful tool for large-scale public health monitoring and intervention (<xref ref-type="bibr" rid="B3">3</xref>).</p>
<p>To address the limitations of early symbolic AI and knowledge-based representations, traditional methods were initially used to analyze EEG data through symbolic reasoning and rule-based systems. These methods attempted to capture and encode domain knowledge, often using rule-based expert systems and symbolic models that were carefully curated by neurologists and psychologists (<xref ref-type="bibr" rid="B4">4</xref>). Such symbolic systems excelled in structured settings, where domain expertise could be meticulously encoded into rule sets for specific, controlled use cases (<xref ref-type="bibr" rid="B5">5</xref>). Despite these advantages, however, they were inherently limited in their scalability and adaptability, especially when applied to diverse EEG data in real-world public health settings. The rigid structure of symbolic systems often failed to generalize across different populations and evolving datasets, as the rules required constant refinement to handle variations in EEG signal patterns, thus constraining their utility in large-scale public health applications.</p>
<p>The evolution toward data-driven approaches and machine learning methods brought more flexibility and data-adaptiveness to EEG analysis. In this stage, researchers utilized traditional machine learning models such as support vector machines (SVMs) (<xref ref-type="bibr" rid="B6">6</xref>), k-nearest neighbors (KNN) (<xref ref-type="bibr" rid="B7">7</xref>), and random forest Hu et al.&#x00027;s (<xref ref-type="bibr" rid="B8">8</xref>), applying these models to extract meaningful features from EEG signals for various health monitoring tasks. Machine learning techniques allowed for greater data-driven adaptability, as models could be trained on specific datasets and applied to classify or detect mental health states or neurological disorders (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B10">10</xref>). Despite this adaptability, these approaches were limited by their reliance on feature engineering, requiring domain expertise to manually design features from raw EEG signals. Consequently, while machine learning methods provided more scalability than symbolic systems, they were still labor-intensive and often struggled with generalizability when exposed to large, heterogeneous datasets common in public health research.</p>
<p>The recent advances in deep learning and the development of pre-trained models like CLIP have further revolutionized the field, enabling more robust, automated feature extraction and interpretation across large, varied EEG datasets. Deep learning models, particularly convolutional neural networks (CNNs) and recurrent neural networks (RNNs), have shown significant promise in EEG interpretation by learning complex, hierarchical patterns directly from raw data. The introduction of multi-scale fusion techniques further improves these models, allowing for the combination of temporal and spatial features across multiple scales of EEG data, thereby enhancing model robustness and interpretability (<xref ref-type="bibr" rid="B11">11</xref>). Despite their effectiveness, these models often face issues related to scalability, particularly when extended to population-level datasets, and they require substantial computational resources (<xref ref-type="bibr" rid="B12">12</xref>). Pre-trained models like CLIP, designed for contrastive learning on image and text data, represent a novel opportunity to bridge this gap by enabling multi-modal fusion of EEG data with language representations, thus opening new possibilities for scalable, interpretable public health applications.</p>
<p>Given the limitations discussed above, PH-CLIP introduces a novel framework that combines multi-scale fusion with contrastive learning to address the scalability, adaptability, and interpretability challenges inherent in EEG analysis. Unlike traditional EEG models that often focus on single-scale features or require extensive feature engineering, PH-CLIP leverages a modified CLIP-based framework to process EEG data, enabling seamless integration of information across multiple spatial, temporal, and cross-modal scales. This multi-scale fusion allows the model to dynamically capture both localized neural patterns and broader spatiotemporal dynamics, providing a more nuanced representation of EEG signals. A key innovation of PH-CLIP lies in its ability to align EEG data with auxiliary modalities, such as contextual or textual information, through contrastive learning. This alignment not only enhances interpretability by linking neural activity to meaningful outcomes but also improves generalizability across diverse populations and datasets. These advancements make PH-CLIP uniquely suited for large-scale, real-world public health applications, where datasets are often heterogeneous and require robust adaptability. By integrating multi-scale fusion and contrastive learning into a unified framework, PH-CLIP transcends the limitations of existing EEG models, such as constrained scalability or limited interpretability. Its design ensures scalability for large datasets, adaptability to diverse populations, and accessibility for public health monitoring systems, paving the way for innovative and effective EEG-based interventions in real-world settings.</p>
<p>The PH-CLIP approach presents the following advantages:</p>
<list list-type="bullet">
<list-item><p>It introduces a new multi-scale fusion module, enabling simultaneous analysis of EEG data across temporal and spatial scales for improved interpretability and robustness.</p></list-item>
<list-item><p>The method demonstrates high adaptability across different scenarios and populations, showing potential for efficient, scalable use in public health applications.</p></list-item>
<list-item><p>Experimental results indicate significant improvements in cross-population generalization and interpretability, making it suitable for diverse public health monitoring needs.</p></list-item>
</list>
</sec>
<sec id="s2">
<title>2 Related work</title>
<sec>
<title>2.1 Contrastive learning for EEG-based health applications</title>
<p>Contrastive learning has emerged as a transformative approach in the development of scalable and robust machine learning models, particularly within healthcare applications leveraging electroencephalography (EEG) data (<xref ref-type="bibr" rid="B10">10</xref>). The core concept of contrastive learning involves learning meaningful representations by differentiating between similar and dissimilar pairs in a given dataset, often under the CLIP (Contrastive Language-Image Pretraining) framework, which aligns multimodal data, such as text and images, in a shared embedding space. Applying CLIP to EEG data for public health monitoring involves unique challenges and adaptations, especially due to the high-dimensional, noise-sensitive, and temporally dynamic nature of EEG signals. Previous studies have shown promising results by adapting contrastive learning frameworks to capture specific health-related patterns, such as identifying biomarkers for neurological disorders, stress levels, sleep stages, or emotional states (<xref ref-type="bibr" rid="B13">13</xref>). The PH-CLIP model introduces scalability within the CLIP framework by incorporating multi-scale fusion techniques, which are essential for effectively capturing the multi-dimensional complexity of EEG data. This approach aligns with prior works that emphasize the necessity of multi-scale data integration, as EEG signals inherently contain features at different temporal resolutions that are relevant for various health indicators (<xref ref-type="bibr" rid="B14">14</xref>). The success of applying contrastive learning on EEG largely depends on effectively capturing both global and local features, and multi-scale fusion facilitates this by integrating features at different resolutions. This capability allows PH-CLIP to generalize across diverse EEG datasets, which is particularly advantageous in public health contexts where EEG-based insights might need to be adapted across varying demographic or health profiles (<xref ref-type="bibr" rid="B15">15</xref>). The use of contrastive learning in EEG-based health applications also requires addressing signal-specific challenges, such as artifact removal, feature extraction, and inter-subject variability. Studies have implemented several preprocessing and augmentation techniques, including Fourier transforms, wavelet decompositions, and other domain-specific transformations, to improve signal fidelity and robustness of learned representations. These methods reduce noise and ensure that the model learns from pertinent patterns, thereby enhancing its scalability across different tasks within public health. The PH-CLIP model&#x00027;s design considers these challenges by integrating multi-scale fusion, thereby aligning with the latest advancements in contrastive learning frameworks for healthcare applications that focus on resilience to data variability and noise while capturing meaningful health signals (<xref ref-type="bibr" rid="B16">16</xref>).</p>
</sec>
<sec>
<title>2.2 Multi-scale fusion techniques for temporal data</title>
<p>Multi-scale fusion techniques have gained significant traction in the context of temporal data analysis, especially where datasets exhibit features across multiple time resolutions. For EEG data, which is characterized by high temporal and frequency dynamics, multi-scale fusion serves as a powerful tool to integrate signals captured at different scales, such as short-term oscillations and long-term trends. This is critical in applications involving public health, where EEG data is analyzed to monitor conditions that manifest across varied temporal resolutions, from immediate stress responses to long-term cognitive decline. Integrating multi-scale fusion with contrastive frameworks like CLIP is particularly promising, as it allows the model to learn representations that retain coherence across these diverse temporal scales (<xref ref-type="bibr" rid="B17">17</xref>). In the case of PH-CLIP, the multi-scale fusion mechanism captures EEG patterns across resolutions, thus enhancing the model&#x00027;s capability to distinguish between meaningful signals and background noise. This technique is achieved by first transforming EEG data into representations at various scales, often through down-sampling or spectral filtering techniques, followed by a hierarchical fusion process that aggregates features into a cohesive representation. Prior studies in other fields, such as speech recognition and activity monitoring, have shown that multi-scale fusion enhances robustness and feature richness by aligning information from short and long time spans. For EEG applications, this method allows for an adaptable model structure that accommodates both immediate and cumulative health indicators (<xref ref-type="bibr" rid="B18">18</xref>). Beyond EEG, multi-scale fusion methods have been explored in other temporal data domains, indicating their broad applicability and utility. Methods such as convolutional and recurrent neural networks, combined with attention mechanisms, have been shown to improve multi-scale processing in fields where complex temporal dependencies exist. In EEG data analysis for public health, these methods enable a more nuanced understanding of brain activity, supporting predictive modeling of health outcomes (<xref ref-type="bibr" rid="B19">19</xref>). PH-CLIP leverages these techniques to address the inherent complexity of EEG signals and aims to improve generalization across diverse health contexts. The integration of multi-scale fusion within PH-CLIP establishes a scalable model architecture that can potentially support applications from individual health monitoring to broader epidemiological studies (<xref ref-type="bibr" rid="B20">20</xref>).</p>
</sec>
<sec>
<title>2.3 Applications of CLIP in public health monitoring</title>
<p>Contrastive learning frameworks, particularly the CLIP model, have shown substantial potential for enhancing public health monitoring by enabling scalable, cross-modal representations of health-related data. The CLIP model&#x00027;s original design aligns textual and visual information, but adaptations for EEG data can allow the alignment of EEG signals with other health-related data types, such as clinical annotations or demographic information. By employing the PH-CLIP model in this context, researchers and public health professionals could leverage a unified model that draws on rich EEG signals to produce health insights relevant to diverse monitoring needs, including early detection of mental health conditions, stress assessment, or cognitive state analysis (<xref ref-type="bibr" rid="B21">21</xref>). Adaptations of CLIP for public health applications have been explored by aligning biomedical signals with complementary data modalities to facilitate comprehensive health assessments. In EEG analysis, this approach can capture associations between brain activity patterns and reported health outcomes, thus enabling a cross-modal understanding of health conditions. The application of such a model in public health has implications for large-scale, non-invasive monitoring systems that could be deployed in community or clinical settings to monitor mental health trends or the impact of environmental stressors on neurological health. PH-CLIP&#x00027;s design, focusing on scalability and multi-scale fusion, enhances this applicability by allowing robust EEG representation across varied health contexts, making it particularly suited for generalizable public health insights (<xref ref-type="bibr" rid="B22">22</xref>). Previous research on public health applications of CLIP has also focused on the challenge of model interpretability, especially given the need for transparent models in health contexts. Adaptations such as explainable AI (XAI) techniques can be integrated with PH-CLIP to interpret which EEG features contribute to specific health predictions, thereby supporting actionable insights for public health interventions. This interpretability is critical in understanding the specific neurological markers that may indicate cognitive decline, emotional stress, or other health states relevant to public health. Overall, PH-CLIP&#x00027;s scalability and interpretability make it a promising approach for deploying CLIP models in public health settings, supporting large-scale and accessible health monitoring systems that adapt to the diverse needs of a population (<xref ref-type="bibr" rid="B23">23</xref>).</p>
</sec>
<sec>
<title>2.4 Advances in multi-modal EEG data integration</title>
<p>Recent years have witnessed significant progress in EEG data analysis, particularly in the integration of multi-modal data to enhance robustness, interpretability, and generalizability. Traditional EEG analysis methods primarily rely on single-modal feature extraction, focusing on temporal or spectral characteristics of neural signals. While these approaches have achieved notable success in applications such as mental health monitoring and cognitive state recognition, they often fail to capture the contextual and environmental factors influencing EEG signals, limiting their applicability in real-world scenarios. One promising direction involves the integration of auxiliary data modalities, such as textual, visual, or physiological signals, to complement EEG analysis. For example, Gaidai and Yihan (<xref ref-type="bibr" rid="B24">24</xref>) proposed a fusion framework combining EEG and eye-tracking data for improved emotion recognition, demonstrating that cross-modal feature alignment enhances model performance in complex tasks. Similarly, Han et al. (<xref ref-type="bibr" rid="B25">25</xref>) utilized speech and EEG data jointly to analyze cognitive workload, highlighting the potential of multi-modal integration to capture subtle interactions between brain activity and external stimuli. Deep learning methods, particularly those leveraging attention mechanisms, have played a pivotal role in advancing multi-modal EEG integration. Transformer-based architectures, as explored in Gaidai et al. (<xref ref-type="bibr" rid="B26">26</xref>), align EEG features with video data to understand affective states in dynamic environments. However, these models often face challenges related to computational efficiency and overfitting in small datasets. Graph Neural Networks (GNNs), such as those used in Qeadan et al. (<xref ref-type="bibr" rid="B27">27</xref>), have also shown promise by modeling spatial dependencies between EEG electrodes and correlating them with auxiliary data sources, such as motion or heart rate. Despite these advances, existing models often struggle with scalability and generalizability across diverse populations and datasets. Many approaches require extensive pre-processing or domain-specific feature engineering, which hinders their deployment in real-world applications. Moreover, multi-modal alignment techniques often lack interpretability, making it difficult to derive actionable insights from the integrated features. To address these limitations, PH-CLIP leverages a contrastive learning framework modified for EEG data, enabling robust alignment of EEG features with auxiliary modalities such as textual or contextual information. By integrating multi-scale fusion with cross-modal learning, PH-CLIP captures the complex interactions between neural activity and external factors, ensuring adaptability and scalability for large-scale public health applications. This positions PH-CLIP as a significant advancement over existing methods, offering a unified framework for EEG-based multi-modal analysis.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Method</title>
<sec>
<title>3.1 Overview</title>
<p>In this section, we present an innovative forecasting framework specifically designed to address challenges in public health prediction. The framework leverages multi-modal data sources, advanced neural architectures, and specialized strategies for bias mitigation and model interpretability. The approach is structured into several key modules, each contributing a specialized component to ensure accurate, resilient, and meaningful forecasting in the context of public health. In the subsequent sections, we first detail the Mathematical Formulation of our forecasting problem. This includes a comprehensive definition of public health parameters, such as epidemiological variables, social determinants, and health system factors, which are represented within a high-dimensional, multi-modal data space. We introduce notation to formalize the relationships between input features, health outcomes, and temporal dependencies, ensuring the setup for a robust predictive modeling approach. Following this foundational setup, we introduce the Proposed Model Architecture. Our model, denoted as HealthNet, combines features from recurrent neural networks (RNNs), transformers, and graph-based layers to capture complex interactions between epidemiological, social, and behavioral variables. The model design is optimized for flexibility across various public health datasets and incorporates mechanisms to handle both structured and unstructured data sources. HealthNet integrates attention mechanisms to prioritize critical features dynamically, allowing for nuanced handling of time-series data in public health applications (as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>).</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Diagram of HealthNet, an advanced multi-modal forecasting framework for public health prediction, integrating neural architectures like RNNs, transformers, and graph-based layers. The model processes data from multiple modalities&#x02014;text, audio, and visual&#x02014;via specialized feature fusion and cross-feature attention mechanisms to capture complex epidemiological and social interactions. Designed for flexibility, HealthNet incorporates attention layers to prioritize key features and applies bias mitigation techniques to address socioeconomic and demographic disparities, ensuring accurate, interpretable public health forecasts.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-12-1520343-g0001.tif"/>
</fig>
<p>We then elaborate on the Prediction Enhancement Strategies, which further bolster our model&#x00027;s capabilities in real-world scenarios. These strategies include domain-specific adjustments, such as debiasing methods to mitigate the impact of socioeconomic and demographic disparities that can skew health outcomes. Additionally, we describe a fine-tuning procedure tailored for adapting the model to varying levels of data sparsity and noise, a common challenge in public health data. Finally, we provide a comprehensive summary of Implementation and Optimization Techniques applied to our model. These techniques encompass parameter tuning, model validation protocols, and specific metrics used to evaluate forecast accuracy in public health contexts. We also discuss interpretability strategies employed to make HealthNet&#x00027;s predictions accessible and actionable for public health professionals and policymakers. Through this structured approach, the proposed framework offers a comprehensive solution for public health forecasting, designed to adapt to diverse datasets and address real-world forecasting challenges effectively.</p>
</sec>
<sec>
<title>3.2 Preliminaries</title>
<p>In this section, we formalize the public health forecasting problem and establish the mathematical notation and structures necessary for the proposed framework. Our objective is to predict specific health outcomes, such as disease incidence, hospitalization rates, or mortality, based on a set of epidemiological, social, and environmental indicators. The challenge involves capturing complex dependencies across time and between different indicators while addressing issues related to data sparsity, heterogeneity, and potential biases inherent in public health data.</p>
<p>Let <inline-formula><mml:math id="M1"><mml:mrow><mml:mi mathvariant="script">X</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> denote a sequence of multi-dimensional input vectors, where <inline-formula><mml:math id="M2"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula> represents the vector of M features observed at time <italic>t</italic>. Each feature <inline-formula><mml:math id="M3"><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> captures information relevant to public health, such as the number of reported disease cases, environmental factors, or sociodemographic characteristics. For each time step <italic>t</italic>, we aim to predict a health outcome variable <italic>y</italic><sub><italic>t</italic></sub>, which could represent disease incidence, hospitalization rates, or other health indicators of interest.</p>
<p>The public health forecasting task can be framed as finding a function <italic>f</italic> that maps the historical data <inline-formula><mml:math id="M4"><mml:mrow><mml:mi mathvariant="script">X</mml:mi></mml:mrow></mml:math></inline-formula> to the predicted outcomes <inline-formula><mml:math id="M5"><mml:mrow><mml:mi mathvariant="script">Y</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> over the observed time horizon <italic>T</italic>. Mathematically, we seek:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mi>L</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>L</italic> is the window size for historical data considered in each prediction. The function <italic>f</italic> is designed to capture temporal dependencies and correlations between different input features, allowing the model to adapt to dynamic changes in public health conditions.</p>
<p>Public health outcomes often exhibit temporal autocorrelation, where past values significantly influence future predictions. We model these dependencies through sequential inputs <bold>x</bold><sub><italic>t</italic>&#x02212;<italic>L</italic>:<italic>t</italic></sub>, where <bold>x</bold><sub><italic>t</italic>&#x02212;<italic>L</italic>:<italic>t</italic></sub> denotes the concatenated vectors {<bold>x</bold><sub><italic>t</italic></sub>, <bold>x</bold><sub><italic>t</italic>&#x02212;1</sub>, &#x02026;, <bold>x</bold><sub><italic>t</italic>&#x02212;<italic>L</italic></sub>}, capturing the recent history of feature observations.</p>
<p>Let <inline-formula><mml:math id="M7"><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>X</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> represent the time series for feature <italic>m</italic> over the entire time horizon. We hypothesize that specific interactions between these features, such as the influence of environmental conditions on disease spread, play a crucial role in accurate forecasting. We therefore define cross-feature dependency functions &#x003D5;<sub><italic>m</italic></sub>, such that:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M9"><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> denotes a transformed representation of feature <italic>m</italic> that incorporates information from other features at time <italic>t</italic>.</p>
<p>To allow the model to adaptively focus on the most relevant time points and features, we introduce an attention mechanism <inline-formula><mml:math id="M10"><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula>, where &#x003C4; denotes a lagged time step relative to <italic>t</italic>. The attention weights <inline-formula><mml:math id="M11"><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> satisfy <inline-formula><mml:math id="M12"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>, and are computed as:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M14"><mml:msubsup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> is a score function that evaluates the relevance of the past time point <italic>t</italic> &#x02212; &#x003C4; for predicting <italic>y</italic><sub><italic>t</italic></sub> based on feature <italic>m</italic>. The output of the attention mechanism, <inline-formula><mml:math id="M15"><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula>, is computed as:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Given the spatial dependencies often observed in public health data, we define an undirected graph <italic>G</italic> &#x0003D; (<italic>V, E</italic>), where <italic>V</italic> represents locations and <italic>E</italic> denotes edges representing relationships. Each node <italic>v</italic> &#x02208; <italic>V</italic> is associated with a vector <bold>x</bold><sub><italic>v</italic></sub> of health indicators specific to that location. The influence of neighboring locations <inline-formula><mml:math id="M17"><mml:mi>u</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:math></inline-formula> on location <italic>v</italic> at time <italic>t</italic> is captured as:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M18"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>w</italic><sub><italic>uv</italic></sub> is a weight capturing the strength of the interaction between nodes <italic>u</italic> and <italic>v</italic>. This graph structure allows us to model region-specific factors and their impacts on local health outcomes, capturing patterns of spatial dependency.</p>
<p>To optimize the forecasting accuracy, we define a prediction loss <inline-formula><mml:math id="M19"><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:math></inline-formula> that measures the error between the predicted and actual outcomes. A common choice is mean squared error (MSE):</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This loss is minimized over the training set to tune the parameters of the function <italic>f</italic>, ensuring that the model learns accurate temporal and feature-based dependencies for public health forecasting.</p>
</sec>
<sec>
<title>3.3 HealthNet model architecture</title>
<p>Our proposed model, termed HealthNet, is designed to capture the intricate temporal, spatial, and cross-feature relationships present in public health data. HealthNet leverages a hybrid architecture combining recurrent neural networks (RNNs), transformer-based attention layers, and graph convolutional networks (GCNs) to account for temporal dependencies, contextual relevance, and spatial correlations, respectively. This section provides a detailed description of HealthNet&#x00027;s components and their integration to achieve accurate and interpretable forecasting in public health settings. The HealthNet model consists of three main modules: Temporal Encoding, Cross-feature Attention Layer, and Graph-based Spatial Aggregation.</p>
<p><bold>Temporal encoding</bold></p>
<p>In our temporal encoding framework, we utilize a multi-layered recurrent neural network (RNN) model to capture the complex sequential dependencies within public health data, where historical observations often impact future outcomes significantly. Each temporal step contains multiple features, and the RNN encodes these into hidden states over time. Let <inline-formula><mml:math id="M21"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> denote the hidden state at time <italic>t</italic> for feature <italic>k</italic>. This state evolves recursively based on the previous hidden state <inline-formula><mml:math id="M22"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> and the current input <inline-formula><mml:math id="M23"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, as formulated:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">RNN</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">RNN</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B3;<sub>RNN</sub> represents the learned parameters of the RNN model. The hidden state <inline-formula><mml:math id="M25"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> captures the temporal dependencies for feature <italic>k</italic> across past intervals, progressively updating to incorporate new information with each time step. By stacking multiple RNN layers, the model can aggregate high-level temporal dependencies, enhancing its capacity to understand long-range patterns (as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>).</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Diagram illustrating a two-stream encoding network for temporal data processing. The framework comprises an EGG signal encoding stream and a temporal encoding stream, each utilizing convolution, max-pooling, and softmax operations to process input signals. These streams feed into an adaptive fusion network that applies multiple transformations, including average-pooling, to integrate information across the channels. This architecture captures both signal-specific and temporal dependencies, facilitating precise, and adaptive temporal encoding for downstream analysis.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-12-1520343-g0002.tif"/>
</fig>
<p>To further optimize feature relevance dynamically, a temporal attention mechanism is introduced, assigning importance scores <inline-formula><mml:math id="M26"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> to the hidden states based on their significance for the prediction target. The attention score calculation integrates an alignment mechanism over hidden states, represented by:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>q</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msup><mml:mo>&#x000B7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">tanh</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Q</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>c</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:munder></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>q</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msup><mml:mo>&#x000B7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">tanh</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Q</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>c</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <bold>Q</bold><sub><italic>z</italic></sub> and <bold>c</bold><sub><italic>z</italic></sub> are learnable matrices and biases, and <bold>q</bold> is a weight vector that projects hidden states into an attention-relevant domain. This mechanism ensures that more informative hidden states contribute proportionally to the model&#x00027;s temporal context.</p>
<p>The temporal context vector, <bold>s</bold><sub><italic>t</italic></sub>, is generated by combining the hidden states with their respective attention scores, effectively summarizing relevant temporal information as:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>s</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <bold>s</bold><sub><italic>t</italic></sub> is then used as input to the prediction layer. This context vector adapts dynamically to the data at each time step, allowing the model to focus on the most critical temporal signals.</p>
<p>To capture both short-term and long-term dependencies effectively, we add a gating mechanism in the RNN hidden state computation. Let <inline-formula><mml:math id="M29"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> denote a forget gate and <inline-formula><mml:math id="M30"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>i</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> an input gate, controlling the memory retention and update dynamics:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M31"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>U</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E11"><label>(11)</label><mml:math id="M32"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>i</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>U</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003C3; is the sigmoid activation, and <bold>W</bold><sub><italic>g</italic></sub>, <bold>W</bold><sub><italic>i</italic></sub>, <bold>U</bold><sub><italic>g</italic></sub>, <bold>U</bold><sub><italic>i</italic></sub>, <bold>b</bold><sub><italic>g</italic></sub>, <bold>b</bold><sub><italic>i</italic></sub> are learnable parameters. The updated hidden state then integrates both the gated prior state and new input information:</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M33"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x02299;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>i</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x02299;</mml:mo><mml:mtext class="textrm" mathvariant="normal">tanh</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>U</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x02299; denotes element-wise multiplication, ensuring that each hidden state incorporates both immediate and historical data. This refined temporal encoding approach enables more robust and context-sensitive predictions.</p>
<p><bold>Cross-feature attention layer</bold></p>
<p>HealthNet incorporates a specialized cross-feature attention layer that dynamically models interdependencies among public health features, allowing the network to account for complex interactions in health-related data. Recognizing that certain features may amplify or attenuate the impact of others, this attention mechanism enhances the model&#x00027;s ability to adjust its focus across features in response to changing conditions over time. By attending to other features, each feature representation can adaptively prioritize influential relationships, improving predictive accuracy.</p>
<p>Let the set of feature embeddings at time <italic>t</italic> be denoted by <inline-formula><mml:math id="M34"><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, where each <inline-formula><mml:math id="M35"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> represents the embedding for feature <italic>m</italic>. The cross-feature attention mechanism calculates a set of attention weights <inline-formula><mml:math id="M36"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, which assign varying levels of influence from feature <italic>m</italic>&#x02032; to feature <italic>m</italic> based on their current contextual relevance. These weights are computed as:</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M37"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x000B7;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02033;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:munder></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x000B7;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mo class="qopname">&#x02033;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where:</p>
<list list-type="bullet">
<list-item><p><inline-formula><mml:math id="M38"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> represents the attention weight quantifying the influence of feature <italic>m</italic>&#x02032; on feature <italic>m</italic> at time <italic>t</italic>.</p></list-item>
<list-item><p><inline-formula><mml:math id="M39"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M40"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> are the embeddings of features <italic>m</italic> and <italic>m</italic>&#x02032; at time <italic>t</italic>, respectively.</p></list-item>
<list-item><p>&#x000B7; denotes the dot product, which measures the similarity between feature embeddings.</p></list-item>
<list-item><p>&#x003C4; &#x0003E; 0 is the temperature parameter that controls the sharpness of the attention distribution:</p>
<list list-type="simple">
<list-item><p>&#x02218; Smaller &#x003C4; values concentrate attention on a few features by amplifying the relative differences in similarity.</p></list-item>
<list-item><p>&#x02218; Larger &#x003C4; values distribute attention more evenly across features by reducing sensitivity to similarity differences.</p></list-item>
</list>
</list-item>
</list>
<p>Once the attention scores <inline-formula><mml:math id="M41"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> are obtained, a cross-feature context vector <inline-formula><mml:math id="M42"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is generated for each feature <italic>m</italic> by summing the influence-weighted embeddings of all other features:</p>
<disp-formula id="E14"><label>(14)</label><mml:math id="M43"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:munder></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This context vector <inline-formula><mml:math id="M44"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> captures the aggregated influence of the other features on feature <italic>m</italic> at the current time step. By integrating <inline-formula><mml:math id="M45"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> with the original embedding <inline-formula><mml:math id="M46"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, we obtain an enhanced feature representation that reflects both the inherent properties of <italic>m</italic> and the dynamically computed influence from other features. This integration is formalized as follows:</p>
<disp-formula id="E15"><label>(15)</label><mml:math id="M47"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>&#x003BB;</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003BB; is a learnable parameter that adjusts the balance between the original feature embedding <inline-formula><mml:math id="M48"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> and the cross-feature context <inline-formula><mml:math id="M49"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>. This weighted sum allows the model to dynamically modulate the influence of cross-feature information according to the context, providing flexibility to focus on relevant interactions without overriding feature-specific details.</p>
<p>Furthermore, an additional self-attention layer can be applied to the enhanced representations <inline-formula><mml:math id="M50"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> to refine the model&#x00027;s focus across features further. This secondary attention mechanism computes an updated representation for each feature that incorporates second-order interdependencies:</p>
<disp-formula id="E16"><label>(16)</label><mml:math id="M51"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msup><mml:mo>&#x000B7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">tanh</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mo class="qopname">&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:munder></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msup><mml:mo>&#x000B7;</mml:mo><mml:mtext class="textrm" mathvariant="normal">tanh</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mo class="qopname">&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <bold>W</bold><sub><italic>h</italic></sub>, <bold>b</bold><sub><italic>h</italic></sub>, and <bold>w</bold> are learnable parameters that govern the attention distribution across enhanced features. The resulting representation leverages the dynamic inter-feature relationships captured by the cross-feature attention, contributing to more context-aware predictions that reflect both temporal dependencies and feature interactions.</p>
<p><bold>Graph-based spatial aggregation</bold></p>
<p>Public health outcomes often exhibit strong spatial correlations influenced by factors such as geographic proximity, population mobility, and shared environmental conditions. For example, infectious diseases may spread across neighboring regions, while socioeconomic factors can lead to similar health outcomes in proximate areas. To effectively capture these spatial dependencies, HealthNet employs a Graph Convolutional Network (GCN) layer that enables information exchange across interconnected locations. The spatial relationships between regions are encoded in a predefined adjacency matrix <bold>A</bold>, where each node represents a geographic area, and edges capture spatial proximity, travel patterns, or other linking factors.</p>
<p>For each geographic location <italic>v</italic>, the GCN layer aggregates information from its neighboring regions <inline-formula><mml:math id="M52"><mml:mrow><mml:mi>u</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> based on the adjacency structure. This aggregation produces a spatial feature vector <bold>s</bold><sub><italic>v,t</italic></sub> at time <italic>t</italic>, which integrates information from nearby locations into a unified representation. The aggregation process is formulated as:</p>
<disp-formula id="E17"><label>(17)</label><mml:math id="M53"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>s</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>A</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>D</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>D</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>u</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <bold>h</bold><sub><italic>u,t</italic></sub> represents the feature embedding of neighboring location <italic>u</italic> at time <italic>t</italic>, <bold>A</bold><sub><italic>uv</italic></sub> specifies the connection strength between locations <italic>u</italic> and <italic>v</italic> in the adjacency matrix, and <bold>D</bold><sub><italic>vv</italic></sub> and <bold>D</bold><sub><italic>uu</italic></sub> are degree matrix entries used to normalize the aggregation. The parameters <bold>W</bold><sub><italic>g</italic></sub> and <bold>b</bold><sub><italic>g</italic></sub> are learnable weights of the GCN, and &#x003C3;(&#x000B7;) is a non-linear activation function such as ReLU or sigmoid. The normalized adjacency structure ensures stability and accounts for varying connectivity across regions.</p>
<p>To capture higher-order spatial dependencies, a multi-hop mechanism extends the aggregation process to include information from more distant regions. For a given location <italic>v</italic>, the spatial feature vector at hop <italic>k</italic>, denoted as <inline-formula><mml:math id="M54"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>s</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, is recursively defined as:</p>
<disp-formula id="E18"><label>(18)</label><mml:math id="M55"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>s</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>A</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>D</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>D</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>u</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>s</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M56"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>s</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> is the initial feature embedding for location <italic>v</italic>, and <inline-formula><mml:math id="M57"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M58"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> are learnable parameters for the <italic>k</italic>-th hop. This recursive process enables the model to aggregate information from both local and distant regions, incorporating broader spatial contexts.</p>
<p>Multi-hop features are combined through a weighted sum to produce the final spatial representation:</p>
<disp-formula id="E19"><label>(19)</label><mml:math id="M59"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>s</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>s</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B1;<sup>(<italic>k</italic>)</sup> are learnable weights that determine the relative importance of each hop. This adaptive weighting mechanism allows the model to balance local and global influences, depending on the spatial structure and the specific task requirements.</p>
<p><bold>Prediction layer and loss function</bold></p>
<p>The final prediction &#x00177;<sub><italic>t</italic></sub> is generated by combining the outputs from the temporal encoding, cross-feature attention, and spatial aggregation layers, each of which contributes unique information to the overall prediction. Specifically, the temporal context vector <bold>c</bold><sub><italic>t</italic></sub> encapsulates temporal dependencies, the cross-feature attention vector <bold>g</bold><sub><italic>t</italic></sub> models inter-feature relationships, and the spatial aggregation vector <bold>s</bold><sub><italic>t</italic></sub> captures spatial correlations across locations. By concatenating these vectors, the prediction layer can access a rich, multi-dimensional representation of the data at time <italic>t</italic>. Formally, the prediction &#x00177;<sub><italic>t</italic></sub> is computed as:</p>
<disp-formula id="E20"><label>(20)</label><mml:math id="M60"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msubsup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>c</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>g</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>s</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where:</p>
<list list-type="bullet">
<list-item><p>&#x00177;<sub><italic>t</italic></sub> represents the predicted health outcome at time <italic>t</italic>.</p></list-item>
<list-item><p><bold>c</bold><sub><italic>t</italic></sub> is the temporal context vector, capturing sequential dependencies and patterns from historical data.</p></list-item>
<list-item><p><bold>g</bold><sub><italic>t</italic></sub> refers to the cross-feature context vector, modeling interdependencies among features and dynamically adapting based on their relationships.</p></list-item>
<list-item><p><bold>s</bold><sub><italic>t</italic></sub> represents the spatial aggregation vector, capturing spatial correlations and influences from neighboring regions.</p></list-item>
<list-item><p>[<bold>c</bold><sub><italic>t</italic></sub>; <bold>g</bold><sub><italic>t</italic></sub>; <bold>s</bold><sub><italic>t</italic></sub>] denotes the concatenation of the vectors, integrating temporal, cross-feature, and spatial information into a unified representation.</p></list-item>
<list-item><p><bold>w</bold><sub><italic>p</italic></sub> is a learnable weight vector in the prediction layer, responsible for mapping the combined features to the predicted outcome.</p></list-item>
<list-item><p><italic>b</italic><sub><italic>p</italic></sub> is a learnable scalar bias term that adjusts the prediction for better fitting.</p></list-item>
</list>
<p>To train HealthNet, we aim to minimize the discrepancy between the predicted values &#x00177;<sub><italic>t</italic></sub> and the true values <italic>y</italic><sub><italic>t</italic></sub> by optimizing the parameters across the entire network. This is accomplished through a mean squared error (MSE) loss function, defined as:</p>
<disp-formula id="E21"><label>(21)</label><mml:math id="M61"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>T</italic> is the total number of time steps, and <italic>y</italic><sub><italic>t</italic></sub> represents the ground truth health outcome at time <italic>t</italic>. The MSE loss penalizes large deviations between predictions and actual values, driving the model to learn accurate patterns in temporal, spatial, and feature-based dependencies.</p>
<p>To enhance model training, we also introduce regularization terms to the loss function, addressing potential issues with overfitting due to the complex, high-dimensional input space. The augmented loss function is expressed as:</p>
<disp-formula id="E22"><label>(22)</label><mml:math id="M62"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BB;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02016;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>w</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msup><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BB;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02016;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msup><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BB;</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02016;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msup><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003BB;<sub>1</sub>, &#x003BB;<sub>2</sub>, and &#x003BB;<sub>3</sub> are regularization coefficients for the prediction weights <bold>w</bold><sub><italic>p</italic></sub>, spatial aggregation weights <bold>W</bold><sub><italic>g</italic></sub>, and cross-feature attention weights <bold>W</bold><sub><italic>h</italic></sub>, respectively. These regularization terms mitigate the risk of overfitting by constraining the magnitude of the weights, promoting a more generalized and stable model.</p>
<p>In addition to MSE, we employ a temporal consistency regularization that encourages stability in predictions across consecutive time steps. This additional regularization term <inline-formula><mml:math id="M63"><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">temp</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> is defined as:</p>
<disp-formula id="E23"><label>(23)</label><mml:math id="M64"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">temp</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M65"><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">temp</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> penalizes abrupt changes in predicted values across time steps, particularly useful in health forecasting where outcomes typically evolve gradually. The final loss function combining MSE, weight regularization, and temporal consistency is:</p>
<disp-formula id="E24"><label>(24)</label><mml:math id="M66"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">total</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">temp</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B1; is a balancing coefficient that controls the influence of the temporal consistency term. This comprehensive loss formulation enables HealthNet to learn from multi-dimensional dependencies while preserving temporal smoothness, leading to more robust and reliable health outcome forecasts.</p>
<p><bold>Multi-scale fusion process</bold></p>
<p>The multi-scale fusion mechanism integrates features across both spatial and temporal dimensions to enhance robustness and interpretability. Temporal features are extracted at multiple resolutions by applying down-sampling and spectral filtering to the raw temporal data, capturing short-term and long-term patterns. A temporal attention mechanism is introduced to assign weights to each resolution dynamically, where the weight for temporal resolution <italic>t</italic> is calculated as:</p>
<disp-formula id="E25"><label>(25)</label><mml:math id="M67"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:munder></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>and <italic>e</italic><sup>(<italic>t</italic>)</sup> is a learnable scoring function. The aggregated temporal features are computed as:</p>
<disp-formula id="E26"><label>(26)</label><mml:math id="M68"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">temporal</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Spatial features are processed using a Graph Convolutional Network (GCN), where each node corresponds to a spatial location and edges represent spatial relationships. The spatial feature at node <italic>v</italic> for layer <italic>k</italic> is given by:</p>
<disp-formula id="E27"><label>(27)</label><mml:math id="M69"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x000C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mi>u</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x000C3;<sub><italic>vu</italic></sub> is the normalized adjacency matrix, <italic>W</italic><sup>(<italic>k</italic>)</sup> is a learnable weight matrix, and &#x003C3; is a non-linear activation function. To capture higher-order dependencies, multi-hop GCN layers are applied, enabling interactions across multiple spatial hops:</p>
<disp-formula id="E28"><label>(28)</label><mml:math id="M70"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mi>u</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>u</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:msubsup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The spatial and temporal representations are combined through an adaptive fusion mechanism, which dynamically adjusts the contributions of each using a learnable parameter &#x003BB;:</p>
<disp-formula id="E29"><label>(29)</label><mml:math id="M71"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">fused</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">temporal</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>&#x003BB;</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">spatial</mml:mtext></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>To further refine the representation, a hierarchical attention mechanism is applied to prioritize specific scales and regions, producing the final integrated feature representation:</p>
<disp-formula id="E30"><label>(30)</label><mml:math id="M72"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">final</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Attention</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">fused</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext class="textrm" mathvariant="normal">context</mml:mtext></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This comprehensive multi-scale fusion process ensures that critical temporal and spatial patterns are effectively captured and integrated, providing a robust representation for downstream tasks.</p>
</sec>
<sec>
<title>3.4 Adaptive contextual adjustment mechanism</title>
<p>In real-world public health forecasting, data quality and availability can vary significantly across different time periods and regions, affecting model reliability. To address these challenges, we propose an Adaptive Contextual Adjustment Mechanism (ACAM) within HealthNet. ACAM is designed to dynamically adjust predictions based on the contextual uncertainty and variability of input features. This mechanism ensures that HealthNet&#x00027;s predictions remain robust even under conditions of sparse, noisy, or incomplete data.</p>
<p><bold>Confidence weighting for feature reliability</bold></p>
<p>In the domain of public health data analysis, certain features consistently demonstrate higher predictive stability, while others may exhibit variability due to external influences or measurement inconsistencies. To address this variability, ACAM incorporates a confidence weighting module that assigns a dynamic reliability score <inline-formula><mml:math id="M73"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> to each feature <italic>m</italic> at time <italic>t</italic>, modulating the influence of each feature according to its reliability. This approach enables the model to prioritize features with stable predictive power, enhancing robustness against noisy or inconsistent data (as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Diagram depicting the confidence-weighting mechanism within the temporal encoding module, designed to enhance feature reliability in public health data. The structure includes key operations for vector stitching, element-wise multiplication, and gating, with components <italic>Z</italic><sub><italic>t</italic></sub> and <italic>R</italic><sub><italic>t</italic></sub> modulating feature influence based on their stability. This architecture ensures adaptive feature weighting, prioritizing reliable features, and dynamically adjusting confidence scores to mitigate the impact of noisy data on downstream processing.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-12-1520343-g0003.tif"/>
</fig>
<p>The confidence score <inline-formula><mml:math id="M74"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> for feature <italic>m</italic> is computed based on both its historical variance and recent deviations, with lower variance indicating higher reliability. This score is calculated as:</p>
<disp-formula id="E31"><label>(31)</label><mml:math id="M75"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003F5;</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M76"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> represents the standard deviation of feature <italic>m</italic> over a defined historical window leading up to time <italic>t</italic>, providing a measure of recent variability. <inline-formula><mml:math id="M77"><mml:mrow><mml:msup><mml:mrow><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> denotes the long-term mean standard deviation of the feature across the entire dataset, offering a baseline measure of its typical variability, and &#x003F5; is a small constant included to prevent division by zero. The exponential function is applied to map the relative stability to a positive confidence score, where higher <inline-formula><mml:math id="M78"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> values reflect more stable and reliable features, while lower values indicate potential unreliability due to higher fluctuations.</p>
<p>These confidence scores are then applied directly to each feature representation <inline-formula><mml:math id="M79"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> within the temporal encoding module, modulating the influence of each feature based on its calculated reliability. The confidence-weighted feature representation <inline-formula><mml:math id="M80"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is formulated as:</p>
<disp-formula id="E32"><label>(32)</label><mml:math id="M81"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x000B7;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M82"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> acts as a scaling factor, effectively down-weighting features that are deemed less reliable, thereby reducing their impact on downstream layers. This weighting mechanism ensures that the temporal encoding and subsequent model components place greater emphasis on consistently reliable features, mitigating the influence of high-variance or noisy inputs.</p>
<p>To further refine the reliability assessment, we introduce a decay mechanism that adjusts <inline-formula><mml:math id="M83"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> over time based on cumulative variance. Define <inline-formula><mml:math id="M84"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> as an exponential moving average of the feature&#x00027;s variance up to time <italic>t</italic>, updated as follows:</p>
<disp-formula id="E33"><label>(33)</label><mml:math id="M85"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:msubsup><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B1; is a smoothing parameter that controls the decay rate. This running average, <inline-formula><mml:math id="M86"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, captures both short-term and long-term variability trends, allowing <inline-formula><mml:math id="M87"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> to be adjusted more responsively to recent changes. We redefine the confidence score <inline-formula><mml:math id="M88"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> to incorporate this decay-adjusted variance as follows:</p>
<disp-formula id="E34"><label>(34)</label><mml:math id="M89"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003F5;</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This decay-based adjustment enables the model to dynamically recalibrate feature confidence, adapting to evolving data patterns in public health outcomes.</p>
<p>The modified confidence-weighted representation <inline-formula><mml:math id="M90"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is then passed to the cross-feature attention layer, where the weighted representations interact with other features. By ensuring that each feature&#x00027;s influence reflects its stability, this approach strengthens the model&#x00027;s resilience to unreliable data, leading to more robust inter-feature interactions and ultimately more reliable predictions.</p>
<p><bold>Self-attention for contextual anomaly detection</bold></p>
<p>In the Adaptive Confidence-Aware Model (ACAM), a self-attention mechanism is utilized to detect and appropriately handle contextual anomalies in the data, which can arise from sudden, unexpected events such as disease outbreaks, natural disasters, or policy changes. These anomalies, characterized by abrupt deviations from typical patterns, may destabilize the model&#x00027;s predictive accuracy if not managed effectively. By integrating anomaly detection within the self-attention mechanism, ACAM can attenuate the influence of anomalous data points, thereby enhancing the model&#x00027;s robustness to irregularities.</p>
<p>At each time step <italic>t</italic>, we compute an anomaly score &#x003B4;<sub><italic>t</italic></sub> that reflects the degree of deviation of the current feature values from their expected values. This score is calculated by averaging the absolute differences between observed feature values <inline-formula><mml:math id="M91"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> and their expected values <inline-formula><mml:math id="M92"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="M93"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is derived from a baseline distribution, such as a moving average or long-term historical trend. The anomaly score is formulated as:</p>
<disp-formula id="E35"><label>(35)</label><mml:math id="M94"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>|</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>M</italic> denotes the number of features. This score &#x003B4;<sub><italic>t</italic></sub> represents the average level of deviation across all features at time <italic>t</italic>; high values indicate the presence of significant anomalies. The expected values <inline-formula><mml:math id="M95"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> can be dynamically updated over time, ensuring that the anomaly detection remains relevant to evolving patterns.</p>
<p>To make the self-attention mechanism anomaly-aware, we integrate the anomaly score &#x003B4;<sub><italic>t</italic></sub> into the calculation of attention weights across time. Specifically, the anomaly score is applied to decay the attention weights for time steps that exhibit high anomaly scores, thus limiting the influence of these potentially unreliable observations on the overall prediction. The attention coefficient <inline-formula><mml:math id="M96"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> for feature <italic>m</italic> between the current time <italic>t</italic> and a previous time <italic>t</italic> &#x02212; &#x003C4; is defined as:</p>
<disp-formula id="E36"><label>(36)</label><mml:math id="M97"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M98"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> represents the raw attention score for feature <italic>m</italic> between <italic>t</italic> and <italic>t</italic> &#x02212; &#x003C4;, and &#x003BB; is a hyperparameter that controls the sensitivity of the model to anomalies. The anomaly-adjusted attention weights <inline-formula><mml:math id="M99"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> decrease as the anomaly score &#x003B4;<sub><italic>t</italic>&#x02212;&#x003C4;</sub> increases, thereby diminishing the contribution of anomalous points in the attention mechanism. This formulation ensures that anomalous values do not exert disproportionate influence on the model&#x00027;s attention-based feature selection and prediction.</p>
<p>To further refine the anomaly influence, we introduce a confidence modulation function <italic>f</italic>(&#x003B4;<sub><italic>t</italic></sub>) that dynamically adjusts the scaling of &#x003B4;<sub><italic>t</italic></sub> based on the anomaly&#x00027;s relative magnitude. We define <italic>f</italic>(&#x003B4;<sub><italic>t</italic></sub>) as follows:</p>
<disp-formula id="E37"><label>(37)</label><mml:math id="M100"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003BA; is a tunable parameter that controls the rate at which the function saturates. For larger anomaly scores &#x003B4;<sub><italic>t</italic></sub>, <italic>f</italic>(&#x003B4;<sub><italic>t</italic></sub>) approaches 1, thereby reducing the corresponding attention weights more significantly. The resulting anomaly-modulated attention weight can now be reformulated as:</p>
<disp-formula id="E38"><label>(38)</label><mml:math id="M101"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>f</italic>(&#x003B4;<sub><italic>t</italic>&#x02212;&#x003C4;</sub>) scales the impact of anomalies, enabling a more nuanced adjustment to attention weights based on the severity of the detected anomaly.</p>
<p>Finally, the reweighted attention scores are applied to generate an anomaly-aware context vector <inline-formula><mml:math id="M102"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>c</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> for each feature <italic>m</italic>, as follows:</p>
<disp-formula id="E39"><label>(39)</label><mml:math id="M103"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>c</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M104"><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is the feature embedding at <italic>t</italic> &#x02212; &#x003C4;. This anomaly-aware context vector effectively downplays the influence of anomalous time points, improving the model&#x00027;s robustness to fluctuations and preserving the integrity of the predictions by focusing on consistent patterns within the data.</p>
<p><bold>Dynamic data imputation for missing entries</bold></p>
<p>Public health datasets frequently contain missing entries due to factors such as incomplete reporting from certain regions, irregular data collection schedules, or unavailable measurements. To address this issue, the Adaptive Confidence-Aware Model (ACAM) integrates a dynamic data imputation layer designed to fill missing values using both spatial information from neighboring regions and temporal trends from historical data. This imputation approach balances the spatial and temporal dependencies to create robust estimates of missing values, minimizing the risk of introducing biases that could distort HealthNet&#x00027;s predictions.</p>
<p>For a missing feature <inline-formula><mml:math id="M105"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> at time <italic>t</italic> in location <italic>v</italic>, the imputed value <inline-formula><mml:math id="M106"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> is computed by combining contributions from spatially proximate regions and recent temporal data. The formulation is as follows:</p>
<disp-formula id="E40"><label>(40)</label><mml:math id="M107"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B1; is a weighting parameter that balances the contributions of spatial and temporal information. The set <inline-formula><mml:math id="M108"><mml:mrow><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula> denotes the neighboring regions of location <italic>v</italic>, with <italic>w</italic><sub><italic>uv</italic></sub> representing the spatial weights that measure the influence of neighboring region <italic>u</italic> on region <italic>v</italic>. The terms &#x003B2;<sub>&#x003C4;</sub> are temporal smoothing coefficients, which define the impact of past time points on the imputation of the current missing value.</p>
<p>The spatial component of the imputation model leverages the adjacency relationships between locations to estimate missing values based on neighboring data. This is especially useful when geographic regions exhibit correlated trends, as in the case of infectious diseases spreading across borders. The spatial weights <italic>w</italic><sub><italic>uv</italic></sub> are derived from the adjacency matrix <bold>A</bold> and normalized by the degree matrix <bold>D</bold>, ensuring that each neighboring region&#x00027;s influence is proportionate to its connection strength. The normalized spatial weight can be defined as:</p>
<disp-formula id="E41"><label>(41)</label><mml:math id="M109"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>A</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>D</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>v</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>D</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>u</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <bold>A</bold><sub><italic>uv</italic></sub> represents the adjacency relation between regions <italic>u</italic> and <italic>v</italic>, and <bold>D</bold><sub><italic>vv</italic></sub> and <bold>D</bold><sub><italic>uu</italic></sub> are the degrees (total connections) of regions <italic>v</italic> and <italic>u</italic>, respectively. This normalization ensures that the spatial contribution to the imputed value is balanced across regions with varying degrees of connectivity.</p>
<p>The temporal component considers historical observations of feature <italic>m</italic> at previous time steps to provide continuity and consistency in imputation. The coefficients &#x003B2;<sub>&#x003C4;</sub> act as temporal smoothing factors, which adjust the influence of each past time step <italic>t</italic> &#x02212; &#x003C4; on the current imputation. These coefficients are chosen to decay over time, reflecting that recent observations are more predictive of the current value than distant ones. A commonly used decay function for temporal smoothing coefficients is the exponential decay:</p>
<disp-formula id="E42"><label>(42)</label><mml:math id="M110"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:msup><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003C1; is a decay rate parameter that controls the rate at which the influence of older data diminishes. The normalization by the sum ensures that <inline-formula><mml:math id="M111"><mml:mrow><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:math></inline-formula>, preserving the overall temporal contribution.</p>
<p>To allow flexibility in the imputation, the weighting parameter &#x003B1; is made adaptive to the data quality and availability from neighboring regions vs. temporal data. For instance, if spatial data is sparse or inconsistent, the model can automatically favor temporal information. An adaptive form of &#x003B1; can be defined based on the relative confidence in spatial and temporal contributions:</p>
<disp-formula id="E43"><label>(43)</label><mml:math id="M112"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>&#x003B1;</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B3;<sub><italic>uv</italic></sub> represents confidence weights for spatial data from neighboring region <italic>u</italic>, and &#x003B7;<sub>&#x003C4;</sub> denotes the reliability of temporal data from time <italic>t</italic> &#x02212; &#x003C4;. Higher values of &#x003B3;<sub><italic>uv</italic></sub> increase &#x003B1;, favoring spatial information, while higher &#x003B7;<sub>&#x003C4;</sub> values reduce &#x003B1;, favoring temporal data.</p>
<p>This dynamic data imputation approach leverages the interplay between spatial and temporal patterns to fill in missing values with minimal bias, thereby enhancing HealthNet&#x00027;s resilience in scenarios with incomplete data. By dynamically adjusting to data availability and stability, the model provides robust imputation for consistent and reliable public health predictions.</p>
<p>To prevent over-reliance on any single feature, particularly in cases of limited or biased data, ACAM includes a feature regularization term in the loss function. This term penalizes disproportionate contributions from individual features and encourages balanced use of all available data:</p>
<disp-formula id="E44"><label>(44)</label><mml:math id="M113"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">reg</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BB;</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">reg</mml:mtext></mml:mrow></mml:msub><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003BB;<sub>reg</sub> is a regularization hyperparameter. By minimizing this term, HealthNet promotes a fair representation of each feature across time, which is particularly valuable in multi-modal and heterogeneous public health datasets.</p>
<p>PH-CLIP introduces several novel aspects that set it apart from traditional machine learning approaches, notably its ability to effectively integrate and process multimodal and multiscale EEG data. PH-CLIP leverages a contrastive learning framework (adapted specifically from CLIP) to align EEG signal representations with text and contextual data. This cross-modal alignment enables PH-CLIP to derive richer, more context-aware embeddings that are often not possible with traditional models that are limited to unimodal feature spaces. Its multi-scale fusion mechanism is tailored for EEG data. The mechanism dynamically combines spatial and temporal features at multiple resolutions, ensuring that both short-term patterns and long-term trends are captured. Unlike traditional approaches that typically require extensive feature engineering to handle this complexity, PH-CLIP seamlessly achieves this integration in its architecture. PH-CLIP also employs hierarchical attention layers to refine the importance of spatial, temporal, and cross-feature representations. This enables the model to adaptively focus on the most relevant features for a given public health task, thereby enhancing its interpretability and robustness. In contrast, traditional models often rely on static feature importance measures, which can limit their flexibility in different scenarios. PH-CLIP is specifically designed to address the scalability challenges of EEG-based models for large-scale public health applications. By combining pre-trained language and image models with a custom multi-scale EEG encoder, PH-CLIP enhances generalization across diverse datasets and populations, overcoming overfitting and domain-specific limitations common in traditional approaches. Together, these innovations position PH-CLIP as a transformative tool in public health that can address the unique challenges posed by multimodal EEG data analysis.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Experimental setup</title>
<sec>
<title>4.1 Dataset</title>
<p>The SST Dataset (<xref ref-type="bibr" rid="B28">28</xref>) is a widely utilized resource in sentiment analysis research, providing annotations for movie reviews that allow for detailed sentiment categorization. Comprising over 10,000 movie review sentences, it supports both binary and fine-grained sentiment classification tasks. The dataset&#x00027;s annotation quality and sentence-level granularity make it a key benchmark for evaluating natural language processing (NLP) models aimed at sentiment understanding, enabling robust training and validation through diverse emotional tones and expressions. The TweetEval Dataset (<xref ref-type="bibr" rid="B29">29</xref>) is a benchmark designed specifically for tweet-based NLP tasks, featuring a unified framework with seven classification tasks including sentiment analysis, offensive language identification, and emoji prediction. This dataset includes over 60,000 annotated tweets, providing a rich source for the development of models tailored to social media text. Its domain-specific characteristics, such as informal language and high noise, make it particularly challenging, driving advancements in robust NLP techniques for real-world social media applications. The ReDial Dataset (<xref ref-type="bibr" rid="B30">30</xref>) is designed for conversational recommendation systems, comprising over 10,000 dialogues where users discuss and recommend movies. This dataset emphasizes human interactions, reflecting realistic conversational contexts that involve user preferences and complex recommendation patterns. It enables research on dialog generation and recommendation accuracy in interactive settings, promoting the development of systems that can effectively understand and respond to user intents within conversational frameworks. The DynaSent Dataset (<xref ref-type="bibr" rid="B31">31</xref>) offers a dynamic sentiment analysis resource, aiming to address limitations in static sentiment datasets by providing continuously updated, challenging examples. With over 50,000 English sentences, it includes annotations on sentiment polarity under diverse linguistic constructions. The dataset is designed to improve model robustness against adversarial inputs, supporting NLP research focused on sentiment classification adaptability across varying contexts and linguistic challenges.</p>
<p>The dataset consists of high-dimensional EEG recordings collected from a diverse population to ensure generalizability in public health applications. Each EEG signal is rigorously preprocessed to mitigate noise and artifacts. Steps include bandpass filtering to retain frequencies relevant to mental health analysis, baseline correction to remove drift, and artifact suppression techniques to account for noise generated by muscle movement or eye blinks. To improve the robustness of the model and prevent overfitting, various data augmentation techniques are applied. These include temporal augmentation, random cropping and window slicing, and spectral perturbations, where specific frequency bands are altered to simulate different signal variations. In addition, Gaussian noise is added to the signal to simulate real-world variations, and channel shuffling is used to introduce changes in spatial dependencies while preserving the overall structure. These preprocessing and augmentation strategies ensure that the model is adaptable to a variety of scenarios, thereby enhancing its ability to generalize to unknown data. This comprehensive dataset preparation approach lays the foundation for the effectiveness of the PH-CLIP framework in EEG-based public health surveillance.</p>
</sec>
<sec>
<title>4.2 Experimental details</title>
<p>Our experiments are conducted using PyTorch on a high-performance computing cluster with NVIDIA A100 GPUs, optimizing computational efficiency and model accuracy. We utilize AdamW as the optimizer with a learning rate of 1 &#x000D7; 10<sup>&#x02212;5</sup>, paired with a cosine learning rate scheduler for fine-tuning models and stabilizing training. Batch size is set to 32 to balance memory usage and convergence speed across all experiments. For evaluation, accuracy, F1-score, and mean reciprocal rank (MRR) serve as primary metrics, allowing comprehensive performance assessment across datasets with distinct task requirements. Each model is pre-trained on relevant corpora, then fine-tuned on each dataset. For the SST dataset, binary and fine-grained classification models are trained over five epochs, allowing optimal sentiment feature extraction. TweetEval models are trained using early stopping based on validation loss due to the high variability and noise inherent in social media texts. ReDial experiments are conducted with transformer-based models such as BERT and GPT, focusing on dialogue coherence and recommendation relevance, while training occurs over ten epochs to allow the models to capture conversational nuances effectively. The DynaSent dataset requires specific attention to adversarial robustness; thus, models undergo training with data augmentation techniques, incorporating paraphrasing and synonym replacement to enhance model resilience. We introduce dropout with a 0.3 probability in all fully connected layers to prevent overfitting, especially given the challenging nature of DynaSent examples. For each model, hyperparameters like dropout rate and learning rate were fine-tuned through grid search to identify optimal configurations across datasets. We employ cross-validation where applicable, splitting each dataset into training, validation, and testing sets at an 80:10:10 ratio. This ensures model generalizability while mitigating dataset-specific biases. We leverage gradient clipping at 1.0 to maintain stable gradients during backpropagation, crucial for handling the complexity of transformer-based models. Each experiment is repeated three times with different random seeds, and the mean results are reported to account for statistical variance in model performance. This setup ensures a rigorous evaluation framework, facilitating detailed performance comparisons across all models and datasets (<xref ref-type="table" rid="T7">Algorithm 1</xref>).</p>
<table-wrap position="float" id="T7">
<label>Algorithm 1</label>
<caption><p>PH-CLIP training procedure on multiple datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-12-1520343-i0001.tif"/>
</table-wrap>
</sec>
<sec>
<title>4.3 Comparison with SOTA methods</title>
<p>The performance of our proposed method is rigorously evaluated against current state-of-the-art (SOTA) models, namely BERT (<xref ref-type="bibr" rid="B32">32</xref>), RoBERTa (<xref ref-type="bibr" rid="B33">33</xref>), ALBERT (<xref ref-type="bibr" rid="B34">34</xref>), ELECTRA (<xref ref-type="bibr" rid="B35">35</xref>), XLNet (<xref ref-type="bibr" rid="B36">36</xref>), and T5 (<xref ref-type="bibr" rid="B37">37</xref>), across the SST, TweetEval, ReDial, and DynaSent datasets, focusing on emotion recognition. As shown in <xref ref-type="table" rid="T1">Tables 1</xref>, <xref ref-type="table" rid="T2">2</xref>, our method outperforms existing models on all key metrics: accuracy, precision, F1-score, and AUC. Notably, our method achieves 92.34% accuracy on the SST dataset, surpassing ELECTRA&#x00027;s previous highest score of 90.14%. For the TweetEval dataset, our approach also leads, achieving 90.10% accuracy, highlighting its adaptability and effectiveness across different textual environments. This superior performance is consistent across the ReDial and DynaSent datasets as well, where our model reaches 90.15 and 91.88% accuracy, respectively, indicating its robustness in handling both structured and conversational text. Our method&#x00027;s advantage can be attributed to several key enhancements in the model architecture and training procedure. Primarily, our design introduces dynamic embeddings that adapt based on context and task requirements, which allows for more precise sentiment detection and emotional nuance recognition. Unlike traditional SOTA models which rely heavily on pre-defined embeddings, our approach recalibrates these representations during fine-tuning, especially on datasets with high variability such as TweetEval and DynaSent. Furthermore, through our integration of adversarial training techniques, our model achieves resilience against perturbations and noise prevalent in datasets like TweetEval and DynaSent. The robustness observed in our model&#x00027;s F1 scores, particularly on the DynaSent dataset (90.78%), demonstrates its effective handling of adversarial inputs that challenge conventional SOTA models, which struggle to maintain performance consistency under similar conditions. Another core innovation lies in the model&#x00027;s use of a hybrid attention mechanism that combines self-attention with cross-layer attention, which enhances its capacity to capture long-range dependencies and contextual coherence. This is especially advantageous for datasets like ReDial, where conversational dynamics demand intricate context retention across dialogue turns. Consequently, our method attains superior scores in accuracy and AUC, reaching an AUC of 91.20% on ReDial, outperforming ELECTRA&#x00027;s 88.33%. By leveraging this advanced attention mechanism, the model is better positioned to identify subtle shifts in tone and sentiment within user interactions, an area where simpler attention mechanisms in SOTA models tend to underperform. Moreover, through our customized training strategy, which includes early stopping based on validation loss and progressive learning rate decay, our model remains robust across varying dataset sizes and structures, ensuring high performance without overfitting.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Comparison of our method with SOTA methods on SST and TweetEval datasets for emotion recognition.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center" colspan="4"><bold>SST dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>TweetEval dataset</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>F1 score</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>F1 score</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">BERT (<xref ref-type="bibr" rid="B32">32</xref>)</td>
<td valign="top" align="center">88.23 &#x000B1; 0.02</td>
<td valign="top" align="center">86.19 &#x000B1; 0.02</td>
<td valign="top" align="center">87.56 &#x000B1; 0.02</td>
<td valign="top" align="center">89.14 &#x000B1; 0.03</td>
<td valign="top" align="center">85.91 &#x000B1; 0.02</td>
<td valign="top" align="center">84.22 &#x000B1; 0.02</td>
<td valign="top" align="center">85.30 &#x000B1; 0.02</td>
<td valign="top" align="center">86.57 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">RoBERTa (<xref ref-type="bibr" rid="B33">33</xref>)</td>
<td valign="top" align="center">89.45 &#x000B1; 0.03</td>
<td valign="top" align="center">87.30 &#x000B1; 0.02</td>
<td valign="top" align="center">88.45 &#x000B1; 0.02</td>
<td valign="top" align="center">90.03 &#x000B1; 0.03</td>
<td valign="top" align="center">86.78 &#x000B1; 0.03</td>
<td valign="top" align="center">85.89 &#x000B1; 0.02</td>
<td valign="top" align="center">86.42 &#x000B1; 0.02</td>
<td valign="top" align="center">87.12 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">ALBERT (<xref ref-type="bibr" rid="B34">34</xref>)</td>
<td valign="top" align="center">87.62 &#x000B1; 0.02</td>
<td valign="top" align="center">85.40 &#x000B1; 0.02</td>
<td valign="top" align="center">86.74 &#x000B1; 0.02</td>
<td valign="top" align="center">88.23 &#x000B1; 0.02</td>
<td valign="top" align="center">84.65 &#x000B1; 0.02</td>
<td valign="top" align="center">83.15 &#x000B1; 0.02</td>
<td valign="top" align="center">84.04 &#x000B1; 0.02</td>
<td valign="top" align="center">85.33 &#x000B1; 0.02</td>
</tr> <tr>
<td valign="top" align="left">ELECTRA (<xref ref-type="bibr" rid="B35">35</xref>)</td>
<td valign="top" align="center">90.14 &#x000B1; 0.02</td>
<td valign="top" align="center">88.78 &#x000B1; 0.02</td>
<td valign="top" align="center">89.34 &#x000B1; 0.02</td>
<td valign="top" align="center">91.45 &#x000B1; 0.03</td>
<td valign="top" align="center">87.20 &#x000B1; 0.02</td>
<td valign="top" align="center">86.35 &#x000B1; 0.02</td>
<td valign="top" align="center">86.87 &#x000B1; 0.02</td>
<td valign="top" align="center">88.27 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">XLNet (<xref ref-type="bibr" rid="B36">36</xref>)</td>
<td valign="top" align="center">88.95 &#x000B1; 0.02</td>
<td valign="top" align="center">87.10 &#x000B1; 0.02</td>
<td valign="top" align="center">87.88 &#x000B1; 0.02</td>
<td valign="top" align="center">89.76 &#x000B1; 0.03</td>
<td valign="top" align="center">85.47 &#x000B1; 0.03</td>
<td valign="top" align="center">84.78 &#x000B1; 0.02</td>
<td valign="top" align="center">85.61 &#x000B1; 0.02</td>
<td valign="top" align="center">86.84 &#x000B1; 0.02</td>
</tr> <tr>
<td valign="top" align="left">T5 (<xref ref-type="bibr" rid="B37">37</xref>)</td>
<td valign="top" align="center">89.13 &#x000B1; 0.02</td>
<td valign="top" align="center">87.50 &#x000B1; 0.02</td>
<td valign="top" align="center">88.23 &#x000B1; 0.02</td>
<td valign="top" align="center">90.57 &#x000B1; 0.03</td>
<td valign="top" align="center">86.13 &#x000B1; 0.02</td>
<td valign="top" align="center">85.21 &#x000B1; 0.02</td>
<td valign="top" align="center">85.89 &#x000B1; 0.02</td>
<td valign="top" align="center">87.09 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>92.34</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>90.12</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>91.25</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>93.01</bold> <bold>&#x000B1;</bold> <bold>0.03</bold></td>
<td valign="top" align="center"><bold>90.10</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>88.67</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>89.43</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>90.56</bold> <bold>&#x000B1;</bold> <bold>0.03</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Results are reported as &#x0201C;mean &#x000B1; standard deviation" based on three independent 5-fold cross-validations. Bold scores indicate statistically significant improvements (<italic>p</italic> &#x0003C; 0.05) over other methods, determined by Student&#x00027;s <italic>t</italic>-test.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Comparison of our method with SOTA methods on ReDial and DynaSent datasets for emotion recognition.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center" colspan="4"><bold>ReDial dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>DynaSent dataset</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>F1 score</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>F1 score</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">BERT (<xref ref-type="bibr" rid="B32">32</xref>)</td>
<td valign="top" align="center">85.47 &#x000B1; 0.02</td>
<td valign="top" align="center">84.20 &#x000B1; 0.02</td>
<td valign="top" align="center">84.95 &#x000B1; 0.02</td>
<td valign="top" align="center">86.63 &#x000B1; 0.03</td>
<td valign="top" align="center">87.12 &#x000B1; 0.02</td>
<td valign="top" align="center">85.89 &#x000B1; 0.02</td>
<td valign="top" align="center">86.23 &#x000B1; 0.02</td>
<td valign="top" align="center">88.04 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">RoBERTa (<xref ref-type="bibr" rid="B33">33</xref>)</td>
<td valign="top" align="center">86.75 &#x000B1; 0.02</td>
<td valign="top" align="center">85.34 &#x000B1; 0.02</td>
<td valign="top" align="center">86.02 &#x000B1; 0.02</td>
<td valign="top" align="center">88.02 &#x000B1; 0.03</td>
<td valign="top" align="center">88.43 &#x000B1; 0.03</td>
<td valign="top" align="center">87.10 &#x000B1; 0.02</td>
<td valign="top" align="center">87.72 &#x000B1; 0.02</td>
<td valign="top" align="center">89.67 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">ALBERT (<xref ref-type="bibr" rid="B34">34</xref>)</td>
<td valign="top" align="center">84.22 &#x000B1; 0.02</td>
<td valign="top" align="center">83.01 &#x000B1; 0.02</td>
<td valign="top" align="center">83.68 &#x000B1; 0.02</td>
<td valign="top" align="center">85.14 &#x000B1; 0.02</td>
<td valign="top" align="center">86.00 &#x000B1; 0.02</td>
<td valign="top" align="center">84.58 &#x000B1; 0.02</td>
<td valign="top" align="center">85.20 &#x000B1; 0.02</td>
<td valign="top" align="center">86.71 &#x000B1; 0.02</td>
</tr> <tr>
<td valign="top" align="left">ELECTRA (<xref ref-type="bibr" rid="B35">35</xref>)</td>
<td valign="top" align="center">87.23 &#x000B1; 0.02</td>
<td valign="top" align="center">86.01 &#x000B1; 0.02</td>
<td valign="top" align="center">86.55 &#x000B1; 0.02</td>
<td valign="top" align="center">88.33 &#x000B1; 0.03</td>
<td valign="top" align="center">89.12 &#x000B1; 0.02</td>
<td valign="top" align="center">87.95 &#x000B1; 0.02</td>
<td valign="top" align="center">88.40 &#x000B1; 0.02</td>
<td valign="top" align="center">90.29 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">XLNet (<xref ref-type="bibr" rid="B36">36</xref>)</td>
<td valign="top" align="center">85.10 &#x000B1; 0.02</td>
<td valign="top" align="center">83.89 &#x000B1; 0.02</td>
<td valign="top" align="center">84.65 &#x000B1; 0.02</td>
<td valign="top" align="center">86.02 &#x000B1; 0.03</td>
<td valign="top" align="center">87.38 &#x000B1; 0.03</td>
<td valign="top" align="center">86.02 &#x000B1; 0.02</td>
<td valign="top" align="center">86.76 &#x000B1; 0.02</td>
<td valign="top" align="center">88.43 &#x000B1; 0.02</td>
</tr> <tr>
<td valign="top" align="left">T5 (<xref ref-type="bibr" rid="B37">37</xref>)</td>
<td valign="top" align="center">86.41 &#x000B1; 0.02</td>
<td valign="top" align="center">85.23 &#x000B1; 0.02</td>
<td valign="top" align="center">85.86 &#x000B1; 0.02</td>
<td valign="top" align="center">87.44 &#x000B1; 0.03</td>
<td valign="top" align="center">88.02 &#x000B1; 0.02</td>
<td valign="top" align="center">86.78 &#x000B1; 0.02</td>
<td valign="top" align="center">87.32 &#x000B1; 0.02</td>
<td valign="top" align="center">89.02 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>90.15</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>88.67</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>89.43</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>91.20</bold> <bold>&#x000B1;</bold> <bold>0.03</bold></td>
<td valign="top" align="center"><bold>91.88</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>90.25</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>90.78</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>92.10</bold> <bold>&#x000B1;</bold> <bold>0.03</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Results are presented as &#x0201C;mean &#x000B1; standard deviation&#x0201D; based on three independent 5-fold cross-validations. Bold values indicate statistically significant improvements (<italic>p</italic> &#x0003C; 0.05) over competing methods, validated through Student&#x00027;s <italic>t</italic>-test.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.4 Ablation study</title>
<p>The ablation study provides an in-depth analysis of our method by systematically removing critical components to evaluate their individual contributions to model performance on the SST, TweetEval, ReDial, and DynaSent datasets. As shown in <xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref> and <xref ref-type="fig" rid="F4">Figures 4</xref>, <xref ref-type="fig" rid="F5">5</xref>, each module plays a significant role in enhancing metrics such as accuracy, precision, F1 score, and AUC, demonstrating the cumulative benefit of these architectural choices. Removing Temporal Encoding led to the most pronounced drop in performance across all datasets, indicating its essential role in facilitating accurate sentiment representation. On the SST dataset, the exclusion of Temporal Encoding results in a 3.22% decrease in accuracy, highlighting the module&#x00027;s impact in detecting nuanced emotional tones. Similar trends are observed on the TweetEval dataset, where the absence of Temporal Encoding leads to a noticeable decline in F1 score (86.20%) compared to the full model (89.43%).</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Ablation study results on our method for emotion recognition across SST and TweetEval datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center" colspan="4"><bold>SST dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>TweetEval dataset</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>F1 score</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>F1 score</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">w/o Temporal Encoding</td>
<td valign="top" align="center">89.12 &#x000B1; 0.02</td>
<td valign="top" align="center">87.20 &#x000B1; 0.02</td>
<td valign="top" align="center">88.15 &#x000B1; 0.02</td>
<td valign="top" align="center">90.33 &#x000B1; 0.03</td>
<td valign="top" align="center">87.00 &#x000B1; 0.02</td>
<td valign="top" align="center">85.78 &#x000B1; 0.02</td>
<td valign="top" align="center">86.20 &#x000B1; 0.02</td>
<td valign="top" align="center">88.12 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">w/o Cross-feature Attention Layer</td>
<td valign="top" align="center">88.45 &#x000B1; 0.02</td>
<td valign="top" align="center">86.57 &#x000B1; 0.02</td>
<td valign="top" align="center">87.23 &#x000B1; 0.02</td>
<td valign="top" align="center">89.78 &#x000B1; 0.03</td>
<td valign="top" align="center">86.23 &#x000B1; 0.02</td>
<td valign="top" align="center">84.89 &#x000B1; 0.02</td>
<td valign="top" align="center">85.43 &#x000B1; 0.02</td>
<td valign="top" align="center">87.44 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">w/o Graph-based Spatial Aggregation</td>
<td valign="top" align="center">87.89 &#x000B1; 0.02</td>
<td valign="top" align="center">85.90 &#x000B1; 0.02</td>
<td valign="top" align="center">86.65 &#x000B1; 0.02</td>
<td valign="top" align="center">89.01 &#x000B1; 0.03</td>
<td valign="top" align="center">85.77 &#x000B1; 0.02</td>
<td valign="top" align="center">84.32 &#x000B1; 0.02</td>
<td valign="top" align="center">84.88 &#x000B1; 0.02</td>
<td valign="top" align="center">86.97 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>92.34</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>90.12</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>91.25</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>93.01</bold> <bold>&#x000B1;</bold> <bold>0.03</bold></td>
<td valign="top" align="center"><bold>90.10</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>88.67</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>89.43</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>90.56</bold> <bold>&#x000B1;</bold> <bold>0.03</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Metrics are reported as &#x0201C;mean &#x000B1; standard deviation&#x0201D; obtained from three separate 5-fold cross-validations. Statistically significant improvements are highlighted in bold, determined using a Student&#x00027;s <italic>t</italic>-test with a significance level of <italic>p</italic> &#x0003C; 0.05.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Ablation study results on our method for emotion recognition across ReDial and DynaSent datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center" colspan="4"><bold>ReDial dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>DynaSent dataset</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>F1 score</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>F1 score</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">w/o Temporal Encoding</td>
<td valign="top" align="center">87.15 &#x000B1; 0.02</td>
<td valign="top" align="center">85.88 &#x000B1; 0.02</td>
<td valign="top" align="center">86.43 &#x000B1; 0.02</td>
<td valign="top" align="center">88.54 &#x000B1; 0.03</td>
<td valign="top" align="center">88.33 &#x000B1; 0.02</td>
<td valign="top" align="center">86.90 &#x000B1; 0.02</td>
<td valign="top" align="center">87.25 &#x000B1; 0.02</td>
<td valign="top" align="center">89.10 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">w/o Cross-feature Attention Layer</td>
<td valign="top" align="center">86.43 &#x000B1; 0.02</td>
<td valign="top" align="center">84.65 &#x000B1; 0.02</td>
<td valign="top" align="center">85.28 &#x000B1; 0.02</td>
<td valign="top" align="center">87.98 &#x000B1; 0.03</td>
<td valign="top" align="center">87.52 &#x000B1; 0.02</td>
<td valign="top" align="center">85.43 &#x000B1; 0.02</td>
<td valign="top" align="center">86.10 &#x000B1; 0.02</td>
<td valign="top" align="center">88.02 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">w/o Graph-based Spatial Aggregation</td>
<td valign="top" align="center">85.98 &#x000B1; 0.02</td>
<td valign="top" align="center">84.12 &#x000B1; 0.02</td>
<td valign="top" align="center">84.77 &#x000B1; 0.02</td>
<td valign="top" align="center">87.23 &#x000B1; 0.03</td>
<td valign="top" align="center">86.89 &#x000B1; 0.02</td>
<td valign="top" align="center">85.10 &#x000B1; 0.02</td>
<td valign="top" align="center">85.78 &#x000B1; 0.02</td>
<td valign="top" align="center">87.67 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>90.15</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>88.67</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>89.43</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>91.20</bold> <bold>&#x000B1;</bold> <bold>0.03</bold></td>
<td valign="top" align="center"><bold>91.88</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>90.25</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>90.78</bold> <bold>&#x000B1;</bold> <bold>0.02</bold></td>
<td valign="top" align="center"><bold>92.10</bold> <bold>&#x000B1;</bold> <bold>0.03</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Results reflect &#x0201C;mean &#x000B1; standard deviation&#x0201D; over three independent 5-fold cross-validations. Bold values represent statistically significant differences (<italic>p</italic> &#x0003C; 0.05), determined by Student&#x00027;s <italic>t</italic>-test.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Performance comparison of SOTA methods on SST and TweetEval datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-12-1520343-g0004.tif"/>
</fig>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Performance comparison of SOTA methods on ReDial and DynaSent datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-12-1520343-g0005.tif"/>
</fig>
<p>Cross-feature Attention Layer, responsible for cross-layer attention integration, contributes significantly to context retention across datasets, particularly on ReDial and DynaSent, where conversational coherence and sentiment shifts require advanced modeling. In its absence, performance on ReDial saw a drop in F1 score by &#x0007E;4%, reinforcing the module&#x00027;s role in capturing multi-turn dialogues&#x00027; contextual intricacies. Without Cross-feature Attention Layer, accuracy also decreases on the TweetEval dataset, showcasing its utility in enhancing attention distribution, crucial for noisy, social media text. This reduction in performance confirms that the hybrid attention mechanism allows our model to maintain consistent context and sentiment identification in dynamically structured datasets. Graph-based Spatial Aggregation, the dynamic embedding adjustment component, although less impactful than Modules A and B, plays a crucial role in ensuring adaptability across varied linguistic contexts. Its removal led to a slight but consistent drop in AUC and F1 scores across datasets, underlining its contribution to robustness against textual diversity, especially in the DynaSent dataset where adversarial variations are frequent. The exclusion of Graph-based Spatial Aggregation resulted in a decline in AUC from 92.10 to 87.67% on DynaSent, suggesting that dynamic embeddings enable the model to better handle adversarial inputs by refining sentiment distinctions. The complete model configuration exhibits the highest scores across all metrics on each dataset, underscoring the collective efficacy of Modules A, B, and C. The integration of these modules not only boosts performance on structured datasets like SST but also ensures resilience and adaptability on complex, context-rich datasets like DynaSent and ReDial. The consistency in F1 score improvements across all datasets in the complete model (91.25% on SST, 89.43% on TweetEval, 89.43% on ReDial, and 90.78% on DynaSent) highlights the balanced contribution of each module. This ablation study substantiates the modular design&#x00027;s effectiveness in supporting robust and precise emotion recognition across diverse text types, proving the superiority of our method over simplified configurations lacking these enhancements.</p>
<p>The <xref ref-type="table" rid="T5">Table 5</xref> and <xref ref-type="fig" rid="F6">Figure 6</xref> showcases a performance comparison between our method and six other state-of-the-art (SOTA) models on two datasets, SST, and TweetEval, across metrics such as Accuracy, Recall, F1 Score, and AUC. The experimental results indicate that our method significantly outperforms the other models on all metrics, particularly excelling in the critical task of emotion recognition. On the SST dataset, our model achieved an Accuracy of 98.01%, which is &#x0007E;1.81% higher than the second-best performing Transformer-EEG, and outperforms by 7.32% in F1 Score. On the TweetEval dataset, our model reached an Accuracy of 98.02%, significantly leading the other models, and also performed best in terms of Recall and AUC, achieving 94.43 and 95.12%, respectively. These results demonstrate that our method not only effectively captures emotion-related features but also shows notable advantages in generalization across datasets and robustness. Additionally, traditional models like EEGNet and DeepConvNet, although performing reasonably well on individual tasks, fall behind our model in composite metrics, particularly when dealing with high-noise social media text data from TweetEval, where the performance gap further widens.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Comparison of our method with SOTA methods on SST and TweetEval datasets for emotion recognition.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center" colspan="4"><bold>SST dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>TweetEval dataset</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
</tr> <tr>
<td valign="top" align="left">DeepConvNet</td>
<td valign="top" align="center">89.7 &#x000B1; 0.01</td>
<td valign="top" align="center">93.08 &#x000B1; 0.02</td>
<td valign="top" align="center">85.59 &#x000B1; 0.02</td>
<td valign="top" align="center">93.52 &#x000B1; 0.03</td>
<td valign="top" align="center">95.48 &#x000B1; 0.02</td>
<td valign="top" align="center">89.93 &#x000B1; 0.03</td>
<td valign="top" align="center">89.05 &#x000B1; 0.02</td>
<td valign="top" align="center">91.8 &#x000B1; 0.01</td>
</tr> <tr>
<td valign="top" align="left">EEGNet</td>
<td valign="top" align="center">89.49 &#x000B1; 0.02</td>
<td valign="top" align="center">89.13 &#x000B1; 0.01</td>
<td valign="top" align="center">86.09 &#x000B1; 0.03</td>
<td valign="top" align="center">89.69 &#x000B1; 0.02</td>
<td valign="top" align="center">89.33 &#x000B1; 0.02</td>
<td valign="top" align="center">91.82 &#x000B1; 0.01</td>
<td valign="top" align="center">89.15 &#x000B1; 0.03</td>
<td valign="top" align="center">90.94 &#x000B1; 0.02</td>
</tr> <tr>
<td valign="top" align="left">RNN-LSTM</td>
<td valign="top" align="center">95.71 &#x000B1; 0.02</td>
<td valign="top" align="center">91.57 &#x000B1; 0.03</td>
<td valign="top" align="center">84.89 &#x000B1; 0.01</td>
<td valign="top" align="center">90.89 &#x000B1; 0.02</td>
<td valign="top" align="center">89.67 &#x000B1; 0.01</td>
<td valign="top" align="center">89.76 &#x000B1; 0.02</td>
<td valign="top" align="center">89.93 &#x000B1; 0.03</td>
<td valign="top" align="center">93.57 &#x000B1; 0.02</td>
</tr> <tr>
<td valign="top" align="left">Transformer-EEG</td>
<td valign="top" align="center">96.2 &#x000B1; 0.03</td>
<td valign="top" align="center">84.55 &#x000B1; 0.01</td>
<td valign="top" align="center">86.8 &#x000B1; 0.02</td>
<td valign="top" align="center">89.91 &#x000B1; 0.01</td>
<td valign="top" align="center">93.5 &#x000B1; 0.02</td>
<td valign="top" align="center">93.15 &#x000B1; 0.03</td>
<td valign="top" align="center">85.8 &#x000B1; 0.02</td>
<td valign="top" align="center">87.7 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">EEG Conformer</td>
<td valign="top" align="center">87.96 &#x000B1; 0.01</td>
<td valign="top" align="center">89.96 &#x000B1; 0.03</td>
<td valign="top" align="center">87.41 &#x000B1; 0.02</td>
<td valign="top" align="center">84.08 &#x000B1; 0.01</td>
<td valign="top" align="center">89.15 &#x000B1; 0.02</td>
<td valign="top" align="center">87.86 &#x000B1; 0.01</td>
<td valign="top" align="center">85.89 &#x000B1; 0.02</td>
<td valign="top" align="center">92.1 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">EEG-Deformer</td>
<td valign="top" align="center">93.22 &#x000B1; 0.03</td>
<td valign="top" align="center">87.49 &#x000B1; 0.02</td>
<td valign="top" align="center">90.23 &#x000B1; 0.01</td>
<td valign="top" align="center">90.58 &#x000B1; 0.02</td>
<td valign="top" align="center">91.46 &#x000B1; 0.03</td>
<td valign="top" align="center">90.84 &#x000B1; 0.01</td>
<td valign="top" align="center">90.38 &#x000B1; 0.02</td>
<td valign="top" align="center">92.46 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center">98.01 &#x000B1; 0.02</td>
<td valign="top" align="center">94.92 &#x000B1; 0.01</td>
<td valign="top" align="center">94.12 &#x000B1; 0.03</td>
<td valign="top" align="center">95.29 &#x000B1; 0.01</td>
<td valign="top" align="center">98.02 &#x000B1; 0.03</td>
<td valign="top" align="center">94.43 &#x000B1; 0.02</td>
<td valign="top" align="center">93.17 &#x000B1; 0.01</td>
<td valign="top" align="center">95.12 &#x000B1; 0.03</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Metrics are expressed as &#x0201C;mean &#x000B1; standard deviation&#x0201D; based on three independent 5-fold cross-validations. Bold scores indicate statistically significant improvements (<italic>p</italic> &#x0003C; 0.05) over other methods, determined via Student&#x00027;s <italic>t</italic>-test.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Ablation study of our method on SST and TweetEval datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-12-1520343-g0006.tif"/>
</fig>
<p>In the <xref ref-type="table" rid="T6">Table 6</xref> and <xref ref-type="fig" rid="F7">Figure 7</xref> we validated the effectiveness of the RNN model (Baseline) in temporal encoding through a set of comparative experiments against six mainstream deep learning models (LSTM, GRU, Transformer, CNN, TCN, and Hybrid CNN-RNN). The experiments spanned two datasets, SST and TweetEval,all reported in the format &#x0201C;mean &#x000B1; standard deviation&#x0201D; to reflect the stability across three 5-fold cross-validations. The results show that the RNN model outperforms the other models on all metrics, particularly notable on the SST dataset where its Accuracy reached 97.07% &#x000B1; 0.03, &#x0007E;1.65% higher than the next best LSTM. In the TweetEval dataset, the RNN model achieved an Accuracy and F1 Score of 97.66% &#x000B1; 0.01 and 92.64% &#x000B1; 0.03, respectively, leading the other methods. Moreover, the RNN model&#x00027;s performance in AUC was especially outstanding, reaching 96.29% &#x000B1; 0.03 and 95.49% &#x000B1; 0.02 on the two datasets respectively, demonstrating its excellent capability in modeling classification boundaries. By contrast, traditional convolutional models (CNN and TCN) showed relatively insufficient performance in feature extraction, and despite good results on some metrics, they fell short in modeling complex temporal dependencies. Models based on the Transformer, due to their higher computational complexity, had limited performance on smaller datasets and failed to fully leverage their long-term sequence modeling advantages. Hybrid models (Hybrid CNN-RNN), while competitive on specific metrics, still lagged in overall performance compared to RNN.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Comparison of model performances on SST and TweetEval datasets with error ranges.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center" colspan="4"><bold>SST dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>TweetEval dataset</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
</tr> <tr>
<td valign="top" align="left">LSTM</td>
<td valign="top" align="center">95.42 &#x000B1; 0.02</td>
<td valign="top" align="center">92.53 &#x000B1; 0.01</td>
<td valign="top" align="center">89.98 &#x000B1; 0.03</td>
<td valign="top" align="center">88.8 &#x000B1; 0.02</td>
<td valign="top" align="center">87.99 &#x000B1; 0.03</td>
<td valign="top" align="center">88.06 &#x000B1; 0.01</td>
<td valign="top" align="center">84.25 &#x000B1; 0.02</td>
<td valign="top" align="center">91.12 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">GRU</td>
<td valign="top" align="center">93.89 &#x000B1; 0.01</td>
<td valign="top" align="center">90.24 &#x000B1; 0.02</td>
<td valign="top" align="center">90.53 &#x000B1; 0.03</td>
<td valign="top" align="center">88.53 &#x000B1; 0.01</td>
<td valign="top" align="center">86.38 &#x000B1; 0.02</td>
<td valign="top" align="center">88.93 &#x000B1; 0.03</td>
<td valign="top" align="center">89.91 &#x000B1; 0.01</td>
<td valign="top" align="center">85.35 &#x000B1; 0.02</td>
</tr> <tr>
<td valign="top" align="left">Transformer</td>
<td valign="top" align="center">88.32 &#x000B1; 0.03</td>
<td valign="top" align="center">91.45 &#x000B1; 0.02</td>
<td valign="top" align="center">86.76 &#x000B1; 0.01</td>
<td valign="top" align="center">86.39 &#x000B1; 0.03</td>
<td valign="top" align="center">88.19 &#x000B1; 0.02</td>
<td valign="top" align="center">88.06 &#x000B1; 0.03</td>
<td valign="top" align="center">88.96 &#x000B1; 0.02</td>
<td valign="top" align="center">87.02 &#x000B1; 0.01</td>
</tr> <tr>
<td valign="top" align="left">CNN</td>
<td valign="top" align="center">91.7 &#x000B1; 0.01</td>
<td valign="top" align="center">87.58 &#x000B1; 0.03</td>
<td valign="top" align="center">83.86 &#x000B1; 0.02</td>
<td valign="top" align="center">87.96 &#x000B1; 0.01</td>
<td valign="top" align="center">93.56 &#x000B1; 0.03</td>
<td valign="top" align="center">85.2 &#x000B1; 0.02</td>
<td valign="top" align="center">86.25 &#x000B1; 0.01</td>
<td valign="top" align="center">93.12 &#x000B1; 0.02</td>
</tr> <tr>
<td valign="top" align="left">TCN</td>
<td valign="top" align="center">86.78 &#x000B1; 0.02</td>
<td valign="top" align="center">91.08 &#x000B1; 0.03</td>
<td valign="top" align="center">85.52 &#x000B1; 0.01</td>
<td valign="top" align="center">85.11 &#x000B1; 0.03</td>
<td valign="top" align="center">89.08 &#x000B1; 0.01</td>
<td valign="top" align="center">87.25 &#x000B1; 0.02</td>
<td valign="top" align="center">87.57 &#x000B1; 0.03</td>
<td valign="top" align="center">92.6 &#x000B1; 0.01</td>
</tr> <tr>
<td valign="top" align="left">Hybrid CNN-RNN</td>
<td valign="top" align="center">87.63 &#x000B1; 0.03</td>
<td valign="top" align="center">90.01 &#x000B1; 0.01</td>
<td valign="top" align="center">88.99 &#x000B1; 0.02</td>
<td valign="top" align="center">87.81 &#x000B1; 0.03</td>
<td valign="top" align="center">95.59 &#x000B1; 0.01</td>
<td valign="top" align="center">89.63 &#x000B1; 0.02</td>
<td valign="top" align="center">86.76 &#x000B1; 0.03</td>
<td valign="top" align="center">86.56 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">RNN (Baseline)</td>
<td valign="top" align="center">97.07 &#x000B1; 0.03</td>
<td valign="top" align="center">95.32 &#x000B1; 0.02</td>
<td valign="top" align="center">94.09 &#x000B1; 0.01</td>
<td valign="top" align="center">96.29 &#x000B1; 0.03</td>
<td valign="top" align="center">97.66 &#x000B1; 0.01</td>
<td valign="top" align="center">94.71 &#x000B1; 0.02</td>
<td valign="top" align="center">92.64 &#x000B1; 0.03</td>
<td valign="top" align="center">95.49 &#x000B1; 0.02</td>
</tr></tbody>
</table>
</table-wrap>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Ablation study of our method on ReDial and DynaSent datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-12-1520343-g0007.tif"/>
</fig>
<p>PH-CLIP represents a significant advancement in EEG-based public health monitoring by addressing key challenges related to cross-population applicability and scalability. Traditional EEG analysis methods often struggle to generalize across diverse populations due to variations in demographic, neurological, and environmental factors. PH-CLIP overcomes these limitations through its multi-scale fusion mechanism, which dynamically integrates spatial and temporal features across resolutions, enabling the model to adapt effectively to heterogeneous data sources. One of PH-CLIP&#x00027;s most impactful contributions is its ability to align multimodal data, such as EEG signals and contextual textual information, using a modified contrastive learning framework. This approach enhances the model&#x00027;s robustness in cross-population scenarios by capturing shared patterns while preserving population-specific nuances. The integration of hierarchical attention layers further supports this adaptability, allowing the model to prioritize features that are most relevant to specific demographic or health contexts, without being overfitted to a particular dataset. Moreover, PH-CLIP&#x00027;s scalable architecture, built upon pre-trained models and tailored for EEG applications, positions it as a transformative tool for large-scale public health initiatives. Its ability to process vast datasets efficiently makes it feasible for real-time monitoring across diverse populations, enabling early detection of mental health issues, neurological conditions, or stress patterns on a global scale. By bridging gaps in generalizability and interpretability, PH-CLIP paves the way for inclusive and effective EEG-based interventions, ultimately contributing to more equitable and precise public health strategies worldwide.</p>
<p>During the development of PH-CLIP, certain assumptions were made to guide the design and implementation of the model. One key assumption is related to the statistical behavior of EEG signals, which are considered to exhibit quasi-stationary properties within specific recording sessions. This means that the statistical characteristics of the signals are expected to remain stable over short time intervals, enabling the extraction of meaningful temporal patterns through the multi-scale fusion mechanism. Additionally, the model assumes that EEG data has undergone standard pre-processing steps, including filtering and artifact removal, to minimize the impact of noise such as muscle movements or eye blinks, which could otherwise obscure important neural information. Another assumption pertains to population variability and data representation. While acknowledging that EEG signals may vary across populations due to demographic and physiological differences, the model presumes that the fundamental neural patterns associated with specific tasks or health outcomes are consistent and transferable. This underpins PH-CLIP&#x00027;s ability to generalize across populations. Furthermore, the model assumes access to well-annotated, balanced, and sufficiently large datasets for supervised training. High-quality labels and data volume are essential for optimizing the contrastive learning framework and achieving robust representation learning. The alignment of EEG data with auxiliary modalities, such as text or contextual information, is also assumed to be temporally consistent to ensure the effectiveness of the multi-modal learning approach. Finally, the computational demands of the PH-CLIP framework assume the availability of high-performance hardware for training and inference. The multi-scale fusion and contrastive learning mechanisms rely on iterative optimization and high-dimensional computations, which may not be feasible on resource-constrained systems. These assumptions highlight the model&#x00027;s current capabilities while pointing to areas for further exploration, such as improving robustness to noisy or imbalanced data, addressing computational constraints, and enhancing generalizability across more diverse populations and datasets. These considerations will inform future work aimed at refining the framework for broader applicability and reliability in real-world settings.</p>
</sec>
</sec>
<sec id="s5">
<title>5 Conclusions and future work</title>
<p>This work introduces PH-CLIP, a scalable and interpretable framework that adapts the Contrastive Language-Image Pretraining (CLIP) architecture for electroencephalogram (EEG) data, specifically addressing the challenges of public health applications. By integrating a novel multi-scale fusion mechanism, PH-CLIP captures the intricate spatiotemporal patterns in EEG signals, enabling robust and granular detection of public health indicators. Experimental results demonstrate that PH-CLIP achieves classification accuracies of 98.01 and 98.02% on the SST and TweetEval datasets, respectively, significantly surpassing state-of-the-art models by up to 5% in accuracy and 7% in F1 Score. Additionally, its hierarchical attention mechanism enhances interpretability, offering insights into critical temporal and spatial features that drive the predictions, an aspect crucial for real-world applications.</p>
<p>Despite these contributions, PH-CLIP faces several limitations that highlight areas for further improvement. First, the computational demands of the multi-scale fusion approach, while effective for capturing complex EEG dynamics, pose challenges for real-time applications on devices with constrained resources. Future work could focus on optimizing computational efficiency through lightweight model architectures or pruning techniques, making PH-CLIP more suitable for deployment in resource-limited environments such as mobile or wearable devices. Second, the current study primarily addresses binary and categorical classifications, which limits the model&#x00027;s applicability to nuanced health states. Extending PH-CLIP to support regression-based outputs or multilabel classifications would enhance its utility for tracking subtle variations in public health indicators over time. Additionally, its adaptability to handle noisy or incomplete datasets remains to be evaluated, as real-world EEG data often suffers from inconsistencies and missing information. Another key area for future research is cross-population generalization. While PH-CLIP demonstrates significant potential for scaling across diverse datasets, further evaluation is needed to ensure robust performance across heterogeneous populations with varying demographic and cultural characteristics. Incorporating advanced domain adaptation techniques could mitigate potential biases and improve model generalizability. Moreover, ensuring ethical deployment, including addressing data privacy concerns and complying with regulatory frameworks such as GDPR, will be critical for real-world public health applications.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>XZ: Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing, Methodology, Supervision, Formal analysis, Project administration, Validation, Investigation, Funding acquisition, Resources, Software. HL: Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing, Data curation, Supervision, Conceptualization, Project administration, Funding acquisition, Visualization. MS: Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. SF: Conceptualization, Methodology, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. The National Social Science Fund of China, Research on the Modernization Process of Ethnic Traditional Sports, 21ATY010.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Islam</surname> <given-names>MA</given-names></name> <name><surname>Hasan</surname> <given-names>M</given-names></name> <name><surname>Tiwari</surname> <given-names>A</given-names></name> <name><surname>Raju</surname> <given-names>MAW</given-names></name> <name><surname>Jannat</surname> <given-names>F</given-names></name> <name><surname>Sangkham</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>Correlation of dengue and meteorological factors in Bangladesh: a public health concern</article-title>. <source>Int J Environ Res Publ Health</source>. (<year>2023</year>) <volume>20</volume>:<fpage>5152</fpage>. <pub-id pub-id-type="doi">10.3390/ijerph20065152</pub-id><pub-id pub-id-type="pmid">36982061</pub-id></citation></ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rich</surname> <given-names>S</given-names></name> <name><surname>Richards</surname> <given-names>V</given-names></name> <name><surname>Mavian</surname> <given-names>C</given-names></name> <name><surname>Switzer</surname> <given-names>W</given-names></name> <name><surname>Magalis</surname> <given-names>BR</given-names></name> <name><surname>Poschman</surname> <given-names>K</given-names></name> <etal/></person-group>. <article-title>Employing molecular phylodynamic methods to identify and forecast HIV transmission clusters in public health settings: a qualitative study</article-title>. <source>Viruses</source>. (<year>2020</year>) <volume>12</volume>:<fpage>921</fpage>. <pub-id pub-id-type="doi">10.3390/v12090921</pub-id><pub-id pub-id-type="pmid">32842636</pub-id></citation></ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Naseem</surname> <given-names>U</given-names></name> <name><surname>Lee</surname> <given-names>BC</given-names></name> <name><surname>Khushi</surname> <given-names>M</given-names></name> <name><surname>Kim</surname> <given-names>J</given-names></name> <name><surname>Dunn</surname> <given-names>A</given-names></name></person-group>. <article-title>Benchmarking for public health surveillance tasks on social media with a domain-specific pretrained language model</article-title>. <source>NLPPOWER</source>. (<year>2022</year>) <volume>27</volume>:<fpage>3</fpage>. <pub-id pub-id-type="doi">10.18653/v1/2022.nlppower-1.3</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dwivedi</surname> <given-names>R</given-names></name> <name><surname>Athe</surname> <given-names>R</given-names></name> <name><surname>Mahesh</surname> <given-names>K</given-names></name> <name><surname>Modem</surname> <given-names>PK</given-names></name></person-group>. <article-title>The incubation period of coronavirus disease (COVID-19): a tremendous public health threat&#x02014;forecasting from publicly available case data in India</article-title>. <source>J Publ Aff</source> . (<year>2021</year>) <volume>2021</volume>:<fpage>2619</fpage>. <pub-id pub-id-type="doi">10.1002/pa.2619</pub-id><pub-id pub-id-type="pmid">33786017</pub-id></citation></ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lucero&#x02013;Prisno</surname> <given-names>D</given-names></name> <name><surname>Kouwenhoven</surname> <given-names>M</given-names></name> <name><surname>Adebisi</surname> <given-names>Y</given-names></name> <name><surname>Miranda</surname> <given-names>A</given-names></name> <name><surname>Gyeltshen</surname> <given-names>D</given-names></name> <name><surname>Suleman</surname> <given-names>MH</given-names></name> <etal/></person-group>. <article-title>Top ten public health challenges to track in 2022</article-title>. <source>Publ Health Challenges</source>. (<year>2022</year>) <volume>1</volume>:<fpage>e21</fpage>. <pub-id pub-id-type="doi">10.1002/puh2.21</pub-id></citation>
</ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Adib</surname> <given-names>K</given-names></name> <name><surname>Hancock</surname> <given-names>P</given-names></name> <name><surname>Rahimli</surname> <given-names>A</given-names></name> <name><surname>Mugisa</surname> <given-names>B</given-names></name> <name><surname>Abdulrazeq</surname> <given-names>F</given-names></name> <name><surname>&#x000C1;guas</surname> <given-names>R</given-names></name> <etal/></person-group>. <article-title>A participatory modelling approach for investigating the spread of COVID-19 in countries of the Eastern Mediterranean Region to support public health decision-making</article-title>. <source>Br Med J Glob Health</source>. (<year>2021</year>) <volume>2021</volume>:<fpage>21251474</fpage>. <pub-id pub-id-type="doi">10.1101/2021.02.10.21251474</pub-id><pub-id pub-id-type="pmid">33762253</pub-id></citation></ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>Q</given-names></name> <name><surname>Guo</surname> <given-names>Y</given-names></name> <name><surname>Wang</surname> <given-names>G</given-names></name> <name><surname>Barnes</surname> <given-names>S</given-names></name></person-group>. <article-title>Big data analytics in the fight against major public health incidents (including COVID-19): a conceptual framework</article-title>. <source>Int J Environ Res Publ Health</source>. (<year>2020</year>) <volume>17</volume>:<fpage>6161</fpage>. <pub-id pub-id-type="doi">10.3390/ijerph17176161</pub-id><pub-id pub-id-type="pmid">32854265</pub-id></citation></ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>Z</given-names></name> <name><surname>Ge</surname> <given-names>Q</given-names></name> <name><surname>Li</surname> <given-names>S</given-names></name> <name><surname>Jin</surname> <given-names>L</given-names></name> <name><surname>Xiong</surname> <given-names>M</given-names></name></person-group>. <article-title>Evaluating the effect of public health intervention on the global-wide spread trajectory of COVID-19</article-title>. <source>medRxiv</source>. (<year>2020</year>). <pub-id pub-id-type="doi">10.1101/2020.03.11.20033639</pub-id></citation>
</ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lauer</surname> <given-names>S</given-names></name> <name><surname>Brown</surname> <given-names>AC</given-names></name> <name><surname>Reich</surname> <given-names>N</given-names></name></person-group>. <article-title>Infectious disease forecasting for public health</article-title>. <source>Popul Biol Vect Borne Dis</source>. (<year>2020</year>) <volume>4</volume>:<fpage>45</fpage>&#x02013;<lpage>68</lpage>. <pub-id pub-id-type="doi">10.1093/oso/9780198853244.003.0004</pub-id></citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Prasad</surname> <given-names>S</given-names></name> <name><surname>Deo</surname> <given-names>R</given-names></name> <name><surname>Downs</surname> <given-names>N</given-names></name> <name><surname>Igoe</surname> <given-names>D</given-names></name> <name><surname>Parisi</surname> <given-names>A</given-names></name> <name><surname>Soar</surname> <given-names>J</given-names></name></person-group>. <article-title>Cloud affected solar UV prediction with three-phase wavelet hybrid convolutional long short-term memory network multi-step forecast system</article-title>. <source>IEEE Access</source>. (<year>2022</year>) <volume>2022</volume>:<fpage>3153475</fpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2022.3153475</pub-id></citation>
</ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gaidai</surname> <given-names>O</given-names></name> <name><surname>Xing</surname> <given-names>Y</given-names></name></person-group>. <article-title>A novel multi regional reliability method for COVID-19 death forecast</article-title>. <source>Eng Sci</source>. (<year>2022</year>) <volume>21</volume>:<fpage>799</fpage>. <pub-id pub-id-type="doi">10.30919/es8d799</pub-id><pub-id pub-id-type="pmid">38116326</pub-id></citation></ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lim</surname> <given-names>JT</given-names></name> <name><surname>Tan</surname> <given-names>KB</given-names></name> <name><surname>Abisheganaden</surname> <given-names>J</given-names></name> <name><surname>Dickens</surname> <given-names>B</given-names></name></person-group>. <article-title>Forecasting upper respiratory tract infection burden using high-dimensional time series data and forecast combinations</article-title>. <source>PLoS Comput Biol</source>. (<year>2023</year>) <volume>19</volume>:<fpage>e1010892</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1010892</pub-id><pub-id pub-id-type="pmid">36749792</pub-id></citation></ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nikooyeh</surname> <given-names>B</given-names></name> <name><surname>Ghodsi</surname> <given-names>D</given-names></name> <name><surname>Amini</surname> <given-names>M</given-names></name> <name><surname>Rabiei</surname> <given-names>S</given-names></name> <name><surname>Rasekhi</surname> <given-names>H</given-names></name> <name><surname>Motlagh</surname> <given-names>ME</given-names></name> <etal/></person-group>. <article-title>Dietary changes during COVID-19 lockdown in Iranian households: are we witnessing a secular trend? a narrative review National Food and Nutrition Surveillance</article-title>. <source>Front Publ Health</source>. (<year>2024</year>) <volume>12</volume>:<fpage>1485423</fpage>. <pub-id pub-id-type="doi">10.3389/fpubh.2024.1485423</pub-id><pub-id pub-id-type="pmid">39525458</pub-id></citation></ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sinclair</surname> <given-names>MR</given-names></name> <name><surname>Ardehali</surname> <given-names>M</given-names></name> <name><surname>Diamantidis</surname> <given-names>CJ</given-names></name> <name><surname>Corsino</surname> <given-names>L</given-names></name></person-group>. <article-title>The diabetes cardiovascular outcomes trials and racial and ethnic minority enrollment: impact, barriers, and potential solutions</article-title>. <source>Front Publ Health</source>. (<year>2024</year>) <volume>12</volume>:<fpage>1412874</fpage>. <pub-id pub-id-type="doi">10.3389/fpubh.2024.1412874</pub-id><pub-id pub-id-type="pmid">39525461</pub-id></citation></ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>P</given-names></name> <name><surname>Wang</surname> <given-names>L</given-names></name> <name><surname>Zhou</surname> <given-names>Y</given-names></name> <name><surname>He</surname> <given-names>J</given-names></name> <name><surname>Zhu</surname> <given-names>B</given-names></name> <name><surname>Wang</surname> <given-names>F</given-names></name> <etal/></person-group>. <article-title>An epidemiological forecast model and software assessing interventions on COVID-19 epidemic in China</article-title>. <source>medRxiv</source>. (<year>2020</year>). <pub-id pub-id-type="doi">10.1101/2020.02.29.20029421</pub-id></citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Papastefanopoulos</surname> <given-names>V</given-names></name> <name><surname>Linardatos</surname> <given-names>P</given-names></name> <name><surname>Kotsiantis</surname> <given-names>S</given-names></name></person-group>. <article-title>COVID-19: a comparison of time series methods to forecast percentage of active cases per population</article-title>. <source>Appl Sci</source>. (<year>2020</year>) <volume>10</volume>:<fpage>3880</fpage>. <pub-id pub-id-type="doi">10.3390/app10113880</pub-id></citation>
</ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nawaz</surname> <given-names>S</given-names></name> <name><surname>Li</surname> <given-names>J</given-names></name> <name><surname>Bhatti</surname> <given-names>U</given-names></name> <name><surname>Bazai</surname> <given-names>S</given-names></name> <name><surname>Zafar</surname> <given-names>A</given-names></name> <name><surname>Bhatti</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>A hybrid approach to forecast the COVID-19 epidemic trend</article-title>. <source>PLoS ONE</source>. (<year>2021</year>) <volume>16</volume>:<fpage>e0256971</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0256971</pub-id><pub-id pub-id-type="pmid">34606503</pub-id></citation></ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>McGough</surname> <given-names>SF</given-names></name> <name><surname>Clemente</surname> <given-names>L</given-names></name> <name><surname>Kutz</surname> <given-names>J</given-names></name> <name><surname>Santillana</surname> <given-names>M</given-names></name></person-group>. <article-title>A dynamic, ensemble learning approach to forecast dengue fever epidemic years in Brazil using weather and population susceptibility cycles</article-title>. <source>J Royal Soc Interf</source> . (<year>2021</year>) <volume>18</volume>:<fpage>20201006</fpage>. <pub-id pub-id-type="doi">10.1098/rsif.2020.1006</pub-id><pub-id pub-id-type="pmid">34129785</pub-id></citation></ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>D</given-names></name> <name><surname>Gui</surname> <given-names>S</given-names></name> <name><surname>Wang</surname> <given-names>X</given-names></name> <name><surname>Wang</surname> <given-names>Q</given-names></name> <name><surname>Qiao</surname> <given-names>J</given-names></name> <name><surname>Tao</surname> <given-names>F</given-names></name> <etal/></person-group>. <article-title>Long-term effects of air pollution on daily outpatient visits for allergic conjunctivitis from 2013 to 2020: a time-series study in Urumqi, China</article-title>. <source>Front Publ Health</source>. (<year>2024</year>) <volume>12</volume>:<fpage>1325956</fpage>. <pub-id pub-id-type="doi">10.3389/fpubh.2024.1325956</pub-id><pub-id pub-id-type="pmid">39525469</pub-id></citation></ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname> <given-names>B</given-names></name> <name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name></person-group>. <article-title>Bibliometric analysis and research trend forecast of healthy urban planning for 40 years (1981&#x02013;2020)</article-title>. <source>Int J Environ Res Publ Health</source>. (<year>2021</year>) <volume>18</volume>:<fpage>9444</fpage>. <pub-id pub-id-type="doi">10.3390/ijerph18189444</pub-id><pub-id pub-id-type="pmid">34574368</pub-id></citation></ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lindner</surname> <given-names>P</given-names></name> <name><surname>Forsstr&#x000F6;m</surname> <given-names>D</given-names></name> <name><surname>Jonsson</surname> <given-names>J</given-names></name> <name><surname>Berman</surname> <given-names>AH</given-names></name> <name><surname>Carlbring</surname> <given-names>P</given-names></name></person-group>. <article-title>Transitioning between online gambling modalities and decrease in total gambling activity, but no indication of increase in problematic online gambling intensity during the first phase of the COVID-19 outbreak in Sweden: a time series forecast study</article-title>. <source>Front Publ Health</source>. (<year>2020</year>) <volume>29</volume>:<fpage>554542</fpage>. <pub-id pub-id-type="doi">10.3389/fpubh.2020.554542</pub-id><pub-id pub-id-type="pmid">33117770</pub-id></citation></ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ibrahim</surname> <given-names>M</given-names></name> <name><surname>Haworth</surname> <given-names>J</given-names></name> <name><surname>Lipani</surname> <given-names>A</given-names></name> <name><surname>Aslam</surname> <given-names>N</given-names></name> <name><surname>Cheng</surname> <given-names>T</given-names></name> <name><surname>Christie</surname> <given-names>N</given-names></name></person-group>. <article-title>Variational-LSTM autoencoder to forecast the spread of coronavirus across the globe</article-title>. <source>medRxiv</source>. (<year>2020</year>). <pub-id pub-id-type="doi">10.1101/2020.04.20.20070938</pub-id><pub-id pub-id-type="pmid">33507932</pub-id></citation></ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Horigian</surname> <given-names>VE</given-names></name> <name><surname>Schmidt</surname> <given-names>RD</given-names></name> <name><surname>Feaster</surname> <given-names>D</given-names></name></person-group>. <article-title>Loneliness, mental health, and substance use among US young adults during COVID-19</article-title>. <source>J Psychoact Drugs</source>. (<year>2020</year>) <volume>53</volume>:<fpage>1</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1080/02791072.2020.1836435</pub-id><pub-id pub-id-type="pmid">33111650</pub-id></citation></ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gaidai</surname> <given-names>O</given-names></name> <name><surname>Yihan</surname> <given-names>Y</given-names></name></person-group>. <article-title>A novel bio-system reliability approach for multi-state COVID-19 epidemic forecast</article-title>. <source>Eng Sci</source>. (<year>2022</year>) <volume>21</volume>:<fpage>es8d797</fpage>. <pub-id pub-id-type="doi">10.30919/es8d797</pub-id></citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>Y</given-names></name> <name><surname>Lam</surname> <given-names>J</given-names></name> <name><surname>Li</surname> <given-names>V</given-names></name> <name><surname>Zhang</surname> <given-names>Q</given-names></name></person-group>. <article-title>A domain-specific Bayesian deep-learning approach for air pollution forecast</article-title>. <source>IEEE Trans Big Data</source>. (<year>2022</year>) <volume>8</volume>:<fpage>1034</fpage>&#x02013;<lpage>46</lpage>. <pub-id pub-id-type="doi">10.1109/TBDATA.2020.3005368</pub-id></citation>
</ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gaidai</surname> <given-names>O</given-names></name> <name><surname>Wang</surname> <given-names>F</given-names></name> <name><surname>Yakimov</surname> <given-names>V</given-names></name></person-group>. <article-title>COVID-19 multi-state epidemic forecast in India</article-title>. <source>Proc Ind Natl Sci Acad</source>. (<year>2023</year>) <volume>89</volume>:<fpage>154</fpage>&#x02013;<lpage>61</lpage>. <pub-id pub-id-type="doi">10.1007/s43538-022-00147-5</pub-id></citation>
</ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qeadan</surname> <given-names>F</given-names></name> <name><surname>Honda</surname> <given-names>T</given-names></name> <name><surname>Gren</surname> <given-names>L</given-names></name> <name><surname>Dailey-Provost</surname> <given-names>J</given-names></name> <name><surname>Benson</surname> <given-names>LS</given-names></name> <name><surname>Vanderslice</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>Naive forecast for COVID-19 in Utah based on the South Korea and Italy models-the fluctuation between two extremes</article-title>. <source>Int J Environ Res Publ Health</source>. (<year>2020</year>) <volume>17</volume>:<fpage>2750</fpage>. <pub-id pub-id-type="doi">10.3390/ijerph17082750</pub-id><pub-id pub-id-type="pmid">32316165</pub-id></citation></ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Socher</surname> <given-names>R</given-names></name> <name><surname>Perelygin</surname> <given-names>A</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Chuang</surname> <given-names>J</given-names></name> <name><surname>Manning</surname> <given-names>CD</given-names></name> <name><surname>Ng</surname> <given-names>AY</given-names></name> <etal/></person-group>. <article-title>Recursive deep models for semantic compositionality over a sentiment treebank</article-title>. In: <source>Proceedings of the 2013 Conference on Empirical Methods in Natural Language Processing</source> (<year>2013</year>). p. <fpage>1631</fpage>&#x02013;<lpage>42</lpage>. Available at: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/D13-1170.pdf">https://aclanthology.org/D13-1170.pdf</ext-link></citation>
</ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barbieri</surname> <given-names>F</given-names></name> <name><surname>Camacho-Collados</surname> <given-names>J</given-names></name> <name><surname>Neves</surname> <given-names>L</given-names></name> <name><surname>Espinosa-Anke</surname> <given-names>L</given-names></name></person-group>. <article-title>Tweeteval: unified benchmark and comparative evaluation for tweet classification</article-title>. <source>arXiv preprint arXiv:201012421</source>. (<year>2020</year>). <pub-id pub-id-type="doi">10.18653/v1/2020.findings-emnlp.148</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>R</given-names></name> <name><surname>Ebrahimi Kahou</surname> <given-names>S</given-names></name> <name><surname>Schulz</surname> <given-names>H</given-names></name> <name><surname>Michalski</surname> <given-names>V</given-names></name> <name><surname>Charlin</surname> <given-names>L</given-names></name> <name><surname>Pal</surname> <given-names>C</given-names></name></person-group>. <article-title>Towards deep conversational recommendations</article-title>. <source>Adv Neural Inform Process Syst</source>. (<year>2018</year>) <volume>31</volume>:<fpage>7617</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1812.07617</pub-id><pub-id pub-id-type="pmid">35733137</pub-id></citation></ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Potts</surname> <given-names>C</given-names></name> <name><surname>Wu</surname> <given-names>Z</given-names></name> <name><surname>Geiger</surname> <given-names>A</given-names></name> <name><surname>Kiela</surname> <given-names>D</given-names></name></person-group>. <article-title>DynaSent: a dynamic benchmark for sentiment analysis</article-title>. <source>arXiv preprint arXiv:201215349</source>. (<year>2020</year>). <pub-id pub-id-type="doi">10.18653/v1/2021.acl-long.186</pub-id><pub-id pub-id-type="pmid">36568019</pub-id></citation></ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Karabila</surname> <given-names>I</given-names></name> <name><surname>Darraz</surname> <given-names>N</given-names></name> <name><surname>EL-Ansari</surname> <given-names>A</given-names></name> <name><surname>Alami</surname> <given-names>N</given-names></name> <name><surname>EL Mallahi</surname> <given-names>M</given-names></name></person-group>. <article-title>BERT-enhanced sentiment analysis for personalized E-commerce recommendations</article-title>. <source>Multimed Tools Appl</source>. (<year>2024</year>) <volume>83</volume>:<fpage>56463</fpage>&#x02013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1007/s11042-023-17689-5</pub-id></citation>
</ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>W</given-names></name> <name><surname>Zeng</surname> <given-names>B</given-names></name> <name><surname>Yin</surname> <given-names>X</given-names></name> <name><surname>Wei</surname> <given-names>P</given-names></name></person-group>. <article-title>An improved aspect-category sentiment analysis model for text sentiment analysis based on RoBERTa</article-title>. <source>Appl Intell</source>. (<year>2021</year>) <volume>51</volume>:<fpage>3522</fpage>&#x02013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.1007/s10489-020-01964-1</pub-id></citation>
</ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Kong</surname> <given-names>L</given-names></name> <name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Kong</surname> <given-names>D</given-names></name></person-group>. <article-title>Multi-grained attention representation with ALBERT for aspect-level sentiment classification</article-title>. <source>IEEE Access</source>. (<year>2021</year>) <volume>9</volume>:<fpage>106703</fpage>&#x02013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2021.3100299</pub-id></citation>
</ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Catelli</surname> <given-names>R</given-names></name> <name><surname>Bevilacqua</surname> <given-names>L</given-names></name> <name><surname>Mariniello</surname> <given-names>N</given-names></name> <name><surname>Di Carlo</surname> <given-names>VS</given-names></name> <name><surname>Magaldi</surname> <given-names>M</given-names></name> <name><surname>Fujita</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>A new Italian Cultural Heritage data set: detecting fake reviews with BERT and ELECTRA leveraging the sentiment</article-title>. <source>IEEE Access</source>. (<year>2023</year>) <volume>11</volume>:<fpage>52214</fpage>&#x02013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2023.3277490</pub-id></citation>
</ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Danyal</surname> <given-names>MM</given-names></name> <name><surname>Khan</surname> <given-names>SS</given-names></name> <name><surname>Khan</surname> <given-names>M</given-names></name> <name><surname>Ullah</surname> <given-names>S</given-names></name> <name><surname>Mehmood</surname> <given-names>F</given-names></name> <name><surname>Ali</surname> <given-names>I</given-names></name></person-group>. <article-title>Proposing sentiment analysis model based on BERT and XLNet for movie reviews</article-title>. <source>Multimed Tools Appl</source>. (<year>2024</year>) <volume>2024</volume>:<fpage>1</fpage>&#x02013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1007/s11042-024-18156-5</pub-id></citation>
</ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bird</surname> <given-names>JJ</given-names></name> <name><surname>Ek&#x000E1;rt</surname> <given-names>A</given-names></name> <name><surname>Faria</surname> <given-names>DR</given-names></name></person-group>. <article-title>Chatbot Interaction with Artificial Intelligence: human data augmentation with T5 and language transformer ensemble for text classification</article-title>. <source>J Ambient Intell Human Comput</source>. (<year>2023</year>) <volume>14</volume>:<fpage>3129</fpage>&#x02013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1007/s12652-021-03439-8</pub-id></citation>
</ref>
</ref-list>
</back>
</article>