<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Environ. Sci.</journal-id>
<journal-title>Frontiers in Environmental Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Environ. Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-665X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1653446</article-id>
<article-id pub-id-type="doi">10.3389/fenvs.2025.1653446</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Environmental Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A CEEMDAN-GNN-transformer hybrid model for air quality index forecasting: a case study of Chang&#x2019;an town, Dongguan, China</article-title>
<alt-title alt-title-type="left-running-head">He et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenvs.2025.1653446">10.3389/fenvs.2025.1653446</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>He</surname>
<given-names>Siyuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2850282/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Yuhao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Peng</surname>
<given-names>Jin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Xu</surname>
<given-names>Dean</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Yu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Management School, Guangdong University of Science and Technology</institution>, <addr-line>Dongguan</addr-line>, <addr-line>Guangdong</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Business, Guangxi Minzu Normal University</institution>, <addr-line>Chongzuo</addr-line>, <addr-line>Guangxi</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1968371/overview">Yuhan Huang</ext-link>, University of Technology Sydney, Australia</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1711689/overview">Hengli Wang</ext-link>, Zhongnan University of Economics and Law, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3146720/overview">Yu Zhang</ext-link>, Hefei University of Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Siyuan He, <email>hesiyuan@gdust.edu.cn</email>; Dean Xu, <email>xudean@gxnun.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>10</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>13</volume>
<elocation-id>1653446</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 He, Liu, Peng, Xu and Wang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>He, Liu, Peng, Xu and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Air pollution has emerged as a pressing global environmental issue, and accurate forecasting plays a critical role in environmental governance and public health protection. This study proposes an enhanced air quality forecasting model based on a hybrid CEEMDAN-GNN-Transformer architecture, and conducts an empirical analysis using data from Chang&#x2019;an Town, Dongguan, China. The proposed model first employs Complete Ensemble Empirical Mode Decomposition with Adaptive Noise (CEEMDAN) to extract multi-scale temporal features and mitigate non-stationary noise in the data. Then, a Graph Neural Network (GNN) is applied to capture the spatial dependencies among various air pollutants. Finally, a Transformer model is utilized to model complex temporal dependencies and improve the capture of long-term trends. The research uses historical air quality monitoring data from 2015 to 2024, including concentrations of PM<sub>2.5</sub>, PM<sub>10</sub>, SO<sub>2</sub>, CO, NO<sub>2</sub>, and O<sub>3</sub> as input features, with the Air Quality Index (AQI) as the prediction target. Model performance is enhanced through ablation studies and hyperparameter tuning, and is compared against several mainstream baseline models. Experimental results demonstrate that the proposed CEEMDAN-GNN-Transformer model outperforms traditional approaches in terms of MAE, MSE, and <italic>R</italic>
<sup>2</sup> metrics, achieving superior prediction accuracy and robustness. This study not only contributes to the theoretical advancement of air quality forecasting methodologies but also provides a more precise predictive tool for environmental management and public health risk prevention, offering significant practical value.</p>
</abstract>
<kwd-group>
<kwd>the air quality index (AQI)</kwd>
<kwd>time series</kwd>
<kwd>CEEMDAN</kwd>
<kwd>GNN</kwd>
<kwd>transformer</kwd>
</kwd-group>
<counts>
<page-count count="19"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Environmental Informatics and Remote Sensing</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Air pollution has emerged as one of the most pressing environmental issues worldwide, significantly impacting human health, ecosystems, and economic development. According to the World Health Organization (WHO), over seven million premature deaths annually are attributed to diseases related to air pollution. The United Nations Environment Programme (UNEP) further emphasizes its multifaceted consequences, ranging from respiratory and cardiovascular disorders to ecological degradation and climate change. Key air pollutants&#x2014;such as fine particulate matter (PM<sub>2.5</sub>, PM<sub>10</sub>), sulfur dioxide (SO<sub>2</sub>), carbon monoxide (CO), nitrogen dioxide (NO<sub>2</sub>), and ozone (O<sub>3</sub>)&#x2014;jointly determine the Air Quality Index (AQI), a composite metric widely used to assess atmospheric pollution levels and associated health risks (<xref ref-type="bibr" rid="B14">Horn and Dasgupta, 2024</xref>).</p>
<p>Although the Chinese government has implemented a series of policies to strengthen air pollution control, such as the <italic>Air Pollution Prevention and Control Action Plan</italic>, the <italic>Three-Year Action Plan to Win the Blue Sky Defense Battle</italic>, and the <italic>Air Quality Improvement Action Plan</italic>, the overall air quality in China has improved in recent years. However, some rapidly industrializing and urbanizing regions still face severe air pollution problems. According to the <italic>2023 Bulletin on China&#x2019;s Ecological and Environmental Status</italic> issued by the Ministry of Ecology and Environment, 121 out of 337 cities at the prefecture-level or above still fail to meet the air quality standards. While the PM<sub>2.5</sub> concentration in the Pearl River Delta region has decreased by 34% compared to 2015, the number of days with ozone (O<sub>3</sub>) exceeding the standard has increased to 12.7%, making it the primary pollutant.</p>
<p>Chang&#x2019;an Town in Dongguan, a typical industrial town in the Pearl River Delta, continues to experience significant air quality issues due to the dual pressures of industrial clustering and high population density. In 2022, the annual average PM<sub>2.5</sub> concentration in Chang&#x2019;an Town was 32&#xa0;&#x3bc;g/m<sup>3</sup>, which is below the national secondary standard (35&#xa0;&#x3bc;g/m<sup>3</sup>) but still significantly higher than the World Health Organization&#x2019;s recommended standard (5&#xa0;&#x3bc;g/m<sup>3</sup>). Additionally, the proportion of days with good air quality was lower than the average for Dongguan (<xref ref-type="bibr" rid="B33">Guangdong Provincial Department of Ecology and Environment, 2023</xref>). Against the backdrop of high population density and intense industrial emissions, developing air quality forecasting models with high accuracy and strong generalization ability is of great practical significance for pollution control and environmental governance decision-making.</p>
<p>High-accuracy forecasting models can provide at least a 72-h warning window for pollution events, which is crucial for environmental governance, public health protection, and policy formulation. The variation in air quality is influenced by multiple factors, exhibiting high non-linearity, non-stationarity, and spatial correlation. Traditional statistical models and some machine learning methods often struggle to simultaneously capture multi-scale features, spatial dependencies, and long-term temporal relationships in complex air quality time series data (<xref ref-type="bibr" rid="B21">Mishra and Gupta, 2024</xref>). With the rapid development of deep learning and time series modeling, forecasting methods based on advanced models such as Transformer and Graph Neural Networks (GNN) have demonstrated superior performance. However, single deep learning models often fail to adequately capture the spatiotemporal relationships and multi-scale characteristics between pollutants, which can hinder prediction accuracy. Therefore, combining and optimizing different models has become a key research direction for improving air quality prediction accuracy. Furthermore, due to the presence of significant noise and complex periodic features in air pollution data, employing signal decomposition methods can effectively separate the time series of pollutants, improving the stability and generalization ability of the forecasting model (<xref ref-type="bibr" rid="B2">Agbehadji and Obagbuwa, 2024</xref>).</p>
<p>To address these challenges, this paper proposes a hybrid air quality forecasting framework based on Complete Ensemble Empirical Mode Decomposition with Adaptive Noise (CEEMDAN), GNN, and Transformer. The framework first applies CEEMDAN to decompose the raw pollutant time series data, reducing noise interference and extracting the primary trend information. Next, GNN is used to model the complex spatiotemporal dependencies among pollutants. Finally, the Transformer model is employed to achieve precise forecasting of AQI by leveraging its strong representation power. This study focuses on Chang&#x2019;an Town, Dongguan, using air pollution data from 2015 to 2024, with pollutants such as PM<sub>2.5</sub>, PM<sub>10</sub>, SO<sub>2</sub>, CO, NO<sub>2</sub>, and O<sub>3</sub> as input features, and AQI as the target variable. Empirical analysis is conducted to validate the effectiveness and superiority of the proposed model.</p>
<p>The main contributions of this study are as follows: A hybrid deep learning architecture integrating CEEMDAN, GNN, and Transformer. This model effectively captures the multi-scale, nonlinear, and spatiotemporal characteristics inherent in air quality time series data. The model is empirically validated using a decade-long dataset of air pollutant concentrations collected from an industrial town with complex environmental conditions, demonstrating its robustness in real-world scenarios. The experimental results show that the proposed framework significantly outperforms baseline models in terms of Mean Absolute Error (MAE), Mean Squared Error (MSE), and the coefficient of determination (<italic>R</italic>
<sup>2</sup>), thereby confirming its superior predictive accuracy.</p>
<p>The remainder of this paper is structured as follows: <xref ref-type="sec" rid="s2">Section 2</xref> presents a literature review. <xref ref-type="sec" rid="s3">Section 3</xref> introduces the model architecture and methodology in detail. <xref ref-type="sec" rid="s4">Section 4</xref> discusses data collection and feature processing. <xref ref-type="sec" rid="s5">Section 5</xref> presents the experimental design and result analysis. <xref ref-type="sec" rid="s6">Section 6</xref> concludes the study and outlines future research directions.</p>
</sec>
<sec id="s2">
<title>2 Literature review</title>
<p>In recent years, with the rapid advancement of data science and artificial intelligence, numerous novel forecasting methodologies have emerged, including deep learning models, GNN, and Transformer-based architectures. These approaches have shown considerable potential in air quality forecasting. This section provides a comprehensive review of prior studies on air quality prediction, with a particular focus on hybrid models that integrate CEEMDAN, GNN, and Transformer techniques.</p>
<p>Air quality forecasting, as a crucial task in environmental science, aims to predict future air quality conditions by considering various influencing factors such as meteorology, traffic, industrial emissions, and socioeconomic activities (<xref ref-type="bibr" rid="B1">Abirami and Chitra, 2021</xref>). Traditional forecasting approaches predominantly rely on statistical models such as Autoregressive Integrated Moving Average (ARIMA) and Seasonal ARIMA (SARIMA). These methods perform reasonably well under stationary conditions. For example, <xref ref-type="bibr" rid="B22">Sharma et al. (2025)</xref> applied ARIMA to predict PM<sub>2.5</sub> concentrations in satellite cities around Delhi, India, demonstrating that while ARIMA can capture temporal trends in stationary series, it struggles with external influencing factors and nonlinearities in the data.</p>
<p>To address these limitations, machine learning and deep learning models have been increasingly adopted. Techniques such as Support Vector Machines (SVM), Random Forests (RF), Adaptive Boosting (AdaBoost), and Extreme Gradient Boosting (XGBoost) have improved predictive performance, yet often fail to fully capture spatiotemporal dependencies in the data (<xref ref-type="bibr" rid="B29">Zaini et al., 2022</xref>). <xref ref-type="bibr" rid="B25">Wang et al. (2023)</xref>, for instance, used XGBoost to improve PM<sub>2.5</sub> predictions, though the model was limited in its ability to model spatial correlations. Similarly, K-Nearest Neighbors (KNN) and Gradient Boosted Decision Trees (GBDT) have shown flexibility and adaptability for short-term prediction tasks, but encounter challenges when handling long-term or spatially coupled data (<xref ref-type="bibr" rid="B20">Liu et al., 2021</xref>).</p>
<p>In the deep learning domain, Long Short-Term Memory (LSTM), Convolutional Neural Networks (CNN), and Gated Recurrent Unit (GRU) models have been widely applied to tasks like air quality prediction, where they excel in modeling long-term sequential data. LSTM and GRU are particularly adept at capturing temporal dependencies, though they often require large amounts of training data and computational resources due to their high complexity (<xref ref-type="bibr" rid="B18">Li et al., 2022</xref>). <xref ref-type="bibr" rid="B16">Kumbalaparambi et al. (2023)</xref> proposed a BiLSTM prediction model that effectively captures time dependencies, though the model&#x2019;s interpretability is weak. CNN, known for its powerful feature extraction ability, has also been widely applied to air quality prediction, especially when dealing with spatial information or extracting local temporal features. Some studies have attempted to combine CNN with Transformer models, using CNN to extract local pattern features and Transformer to model long-range dependencies, thus balancing short-term feature capture with long-term trend modeling (<xref ref-type="bibr" rid="B6">Chen et al., 2022</xref>).</p>
<p>Empirical Mode Decomposition (EMD) and its improved version, Complete Ensemble Empirical Mode Decomposition with Adaptive Noise (CEEMDAN), have been widely used in time series data denoising and signal decomposition. CEEMDAN decomposes complex time series signals into multiple intrinsic mode functions (IMFs), each representing a component of the data in a specific frequency band, effectively removing noise while preserving the key features of the signal (<xref ref-type="bibr" rid="B3">Ameri et al., 2023</xref>). CEEMDAN has been applied primarily in the data preprocessing stage of air quality prediction by decomposing different frequency components to optimize the model inputs and enhance prediction accuracy. <xref ref-type="bibr" rid="B27">Wu et al. (2024)</xref> utilized CEEMDAN to decompose PM<sub>2.5</sub> sequences in air quality prediction, successfully removing high-frequency noise and improving model stability. Studies have shown that combining CEEMDAN with deep learning models such as LSTM and GRU can effectively remove noise and short-term fluctuations, thereby enhancing the model&#x2019;s ability to capture long-term trends and improving prediction accuracy (<xref ref-type="bibr" rid="B26">Wu et al., 2022</xref>). <xref ref-type="bibr" rid="B31">Zhang et al. (2023)</xref> proposed a CEEMDAN-LSTM hybrid model for air quality prediction, demonstrating the effectiveness of this approach in noise reduction and improving model accuracy.</p>
<p>The spatiotemporal distribution characteristics of air quality pose challenges for traditional time series models in handling spatial dependencies. Graph Neural Networks (GNN) have been widely applied in air quality prediction due to their ability to capture spatial dependencies by constructing graph structures and learning relationships between nodes. GNN models in air quality prediction typically model the propagation and diffusion of pollutants based on the spatial correlations between monitoring stations (<xref ref-type="bibr" rid="B7">Chen et al., 2023</xref>). <xref ref-type="bibr" rid="B11">Ge et al. (2021)</xref> employed Graph Convolutional Networks (GCN) for urban air quality prediction, successfully incorporating the spatial correlations between monitoring stations to improve prediction accuracy. GNN can learn the diffusion paths of pollutants across different regions, providing more accurate spatial predictions.</p>
<p>The Transformer model, proposed in 2017, has made significant breakthroughs in various fields due to its self-attention mechanism, which effectively handles dependencies in long-time series data. Unlike traditional Recurrent Neural Networks (RNN) and LSTM models, Transformer does not rely on sequential processing and can compute in parallel, significantly improving computational efficiency (<xref ref-type="bibr" rid="B30">Zhang and Zhang, 2023</xref>). In air quality prediction, Transformer and its variants (e.g., Informer, Longformer) have shown their ability to capture complex dependencies in long-time series through self-attention mechanisms. <xref ref-type="bibr" rid="B19">Liang et al. (2023)</xref> proposed a Transformer-based time series prediction model specifically for handling long-term dependencies in multivariate time series data, achieving good predictive performance. Transformer models, by processing large amounts of historical data, are able to better capture the long-term trends in air quality variations, supporting long-term air quality forecasting. <xref ref-type="bibr" rid="B13">He et al. (2025)</xref> proposed a Transformer-based air pollution prediction model that leverages the self-attention mechanism to enhance long-term time dependency modeling.</p>
<p>Hybrid models that combine multiple methods have demonstrated improved performance by leveraging the strengths of different algorithms. CEEMDAN can be integrated with deep learning models like LSTM and GRU to first perform noise reduction and signal decomposition, followed by temporal modeling, thereby enhancing both robustness and accuracy. <xref ref-type="bibr" rid="B23">Tang et al. (2024)</xref> proposed a CEEMDAN-SE-GRU model for air quality forecasting, which effectively captured AQI variations by addressing noise and nonlinear components. Models combining GNN and Transformers can simultaneously process temporal sequences and spatial dependencies, enabling more accurate and context-aware forecasts. For instance, <xref ref-type="bibr" rid="B4">Ban and Shen (2022)</xref> proposed a CEEMDAN&#x2013;LSTM&#x2013;BP&#x2013;ARIMA hybrid model that demonstrated strong adaptability for short-term PM<sub>2.5</sub> prediction.</p>
<p>In summary, current research in air quality prediction exhibits the following trends: (1)Traditional statistical models such as ARIMA and SARIMA remain valuable for stationary, univariate prediction tasks but underperform in nonlinear, multivariate contexts; (2) Machine learning models outperform statistical methods in feature interaction modeling but struggle to jointly capture spatial dependencies and multi-scale temporal features; (3) Deep learning models each possess distinct strengths but often cannot simultaneously address multi-scale temporal structures and spatial correlations within a single architecture; (4) CEEMDAN&#x2013;deep learning hybrids enhance noise reduction and trend extraction but are typically confined to single prediction frameworks, lacking integration of spatial and long-range temporal modeling; (5) GNN&#x2013;Transformer integrations have been applied in other fields but remain underexplored for highly volatile industrial air pollution scenarios.</p>
<p>Accordingly, three major research gaps can be identified: (1) The absence of a unified framework integrating multi-scale signal decomposition, spatial dependency modeling, and long-range temporal dependency capture; (2) Limited empirical validation in typical high-pollution, highly non-stationary industrial zones, restricting the assessment of model stability and generalization; (3) Insufficient fusion between long-range temporal and spatial diffusion modeling, hindering comprehensive characterization of pollutant evolution mechanisms.</p>
<p>Air quality forecasting is a highly complex, multi-factor task involving the nonlinear nature of time series data, spatial dependencies, and the integration of heterogeneous data sources. Although substantial progress has been made in recent years, existing predictive models often struggle to simultaneously capture the nonlinearities, multi-scale temporal structures, and spatial correlations inherent in pollution data. This limitation becomes particularly pronounced in densely populated, heavily industrialized regions such as Chang&#x2019;an Town in Dongguan, China, where pollutant emissions exhibit strong source heterogeneity, abrupt fluctuations, and pronounced spatiotemporal coupling&#x2014;posing considerable challenges to conventional forecasting approaches.</p>
<p>To address these challenges, this study proposes a novel hybrid deep learning framework that integrates CEEMDAN, GNN, and Transformer architectures to jointly model multi-scale structures, spatial diffusion patterns, and long-term temporal dependencies in air quality data. CEEMDAN effectively handles the non-stationarity of pollution time series by decomposing them into intrinsic mode functions across multiple frequency bands, thereby enhancing noise reduction and feature representation. GNN is employed to capture the spatial propagation of pollutants among monitoring stations, uncovering complex regional diffusion relationships. Meanwhile, the Transformer model leverages self-attention mechanisms to strengthen the representation of long-range dependencies and complex temporal dynamics.</p>
<p>This integrated framework overcomes the limitations of single-model approaches in jointly modeling spatial and temporal features, and is particularly well-suited to real-world conditions in industrial zones like Chang&#x2019;an Town, where pollutant concentrations are highly volatile and data are frequently contaminated by noise. By introducing this innovative architecture, the study offers a more adaptive and robust solution for air quality forecasting in environments characterized by strong nonlinearity and spatiotemporal heterogeneity. It provides a scientifically grounded, data-driven tool to support environmental monitoring, early warning systems, and policy-making in complex industrial settings.</p>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methods</title>
<sec id="s3-1">
<title>3.1 CEEMDAN for signal decomposition</title>
<p>The Complete Ensemble Empirical Mode Decomposition with Adaptive Noise (CEEMDAN) is an advanced signal processing method designed for analyzing nonlinear and non-stationary time series. It extends the original Empirical Mode Decomposition (EMD) by improving the decomposition stability and mitigating mode mixing&#x2014;an issue in traditional EMD where intrinsic mode functions (IMFs) with overlapping frequency content compromise the interpretability of results (<xref ref-type="bibr" rid="B12">Guo et al., 2023</xref>).</p>
<p>CEEMDAN enhances EMD by introducing Gaussian white noise into the decomposition process. Initially, Gaussian noise is added to the original signal to generate multiple noisy versions. Each noisy signal is decomposed using EMD to extract its IMFs. Subsequently, the corresponding IMFs from each realization are averaged to obtain a more robust and stable decomposition. The incorporation of adaptive noise further ensures that each IMF retains high independence and spectral purity, thereby improving decomposition accuracy. This process enables effective noise reduction while preserving the underlying characteristics of the original signal (<xref ref-type="bibr" rid="B15">Hu et al., 2021</xref>). In this study, the IMFs and residual components extracted via CEEMDAN serve as refined input features for subsequent spatial-temporal modeling.</p>
</sec>
<sec id="s3-2">
<title>3.2 GNN for spatial feature extraction</title>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> illustrates the basic workflow of Graph Neural Networks (GNN), whose main objective is to extract spatial features from multivariate data at each time step and model the dependencies between variables. Specifically, a graph structure is used to represent the relationships between variables:</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Graph neural network workflow.</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g001.tif">
<alt-text content-type="machine-generated">Diagram of a neural network structure with an input layer, followed by two hidden layers, each with a ReLU activation function, and ending with an output layer. Shapes resembling interconnected stars are depicted within each layer.</alt-text>
</graphic>
</fig>
<p>Nodes represent the 18 variables, including AQI and major pollutant concentrations (e.g., PM<sub>2.5</sub>, PM<sub>10</sub>, SO<sub>2</sub>, CO, NO<sub>2</sub>, O<sub>3</sub>_8&#xa0;h), as well as the IMFs (IMF_1 to IMF_10) and Residue from CEEMDAN decomposition.</p>
<p>Edges define the connections between the variables through an adjacency matrix.</p>
<p>Graph Convolutional Networks (GCN), a common variant of GNN, are used to update the spatial features, and the update process can be represented by the following equations, as shown in <xref ref-type="disp-formula" rid="e1">Equations 1</xref>&#x2013;<xref ref-type="disp-formula" rid="e3">3</xref> (<xref ref-type="bibr" rid="B32">Zhou et al., 2020</xref>):<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mi>W</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the input feature at time <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the normalized adjacency matrix<italic>.</italic> <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the learnable weight matrix, where <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mi mathvariant="normal">D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the hidden layer dimension. <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mi>D</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the bias term. <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the activation function, representing spatial features at time <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>To enhance expressiveness, multiple layers can be stacked, allowing the network to capture more complex relationships.<disp-formula id="e2">
<mml:math id="m10">
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<inline-formula id="inf9">
<mml:math id="m11">
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the initial input, and <inline-formula id="inf10">
<mml:math id="m12">
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the output of the <inline-formula id="inf11">
<mml:math id="m13">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> th layer, serving as the final spatial feature. The GNN is applied across all time steps, producing the output as:<disp-formula id="e3">
<mml:math id="m14">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>T</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>This process is applied at every time step, yielding spatial features across time.</p>
</sec>
<sec id="s3-3">
<title>3.3 Feature fusion mechanism</title>
<p>The spatial features extracted by GNN need to be fused with the original input features to generate a unified input for the Transformer model.</p>
<p>To fully leverage the spatial feature extraction capabilities of GNN, the spatial features are fused with the original input features to form a combined feature suitable for Transformer modeling. The fusion process is carried out as shown in <xref ref-type="disp-formula" rid="e4">Equations 4</xref>&#x2013;<xref ref-type="disp-formula" rid="e6">6</xref>:</p>
<p>Concatenation: The features are concatenated.<disp-formula id="e4">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>Linear Transformation: A linear transformation is applied.<disp-formula id="e5">
<mml:math id="m19">
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>Attention Fusion: Multi-head attention mechanisms dynamically fuse the features.<disp-formula id="e6">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>M</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf15">
<mml:math id="m21">
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (queries) are derived from original features, and <inline-formula id="inf16">
<mml:math id="m22">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> ((keys) and <inline-formula id="inf17">
<mml:math id="m23">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (values) are obtained from GNN features. The output <inline-formula id="inf18">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mo>`</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is reshaped into <inline-formula id="inf19">
<mml:math id="m25">
<mml:mrow>
<mml:mi mathvariant="normal">Z</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mo>`</mml:mo>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, facilitating dynamic weighted fusion and improving sensitivity to key variables.</p>
</sec>
<sec id="s3-4">
<title>3.4 Transformer for temporal modeling</title>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> illustrates the main structure of the Transformer model. The Transformer model receives the fused feature <inline-formula id="inf20">
<mml:math id="m26">
<mml:mrow>
<mml:mi mathvariant="normal">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which is used to model the temporal dependencies in the time series for future prediction. The input processing first reshapes <inline-formula id="inf21">
<mml:math id="m27">
<mml:mrow>
<mml:mi mathvariant="normal">Z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to match the input format required by the Transformer. Then, position encoding is applied to add temporal information to the input sequence, as Transformer models do not inherently account for sequence order (<xref ref-type="bibr" rid="B24">Vaswani et al., 2017</xref>). Position encoding typically uses sinusoidal functions, as shown in <xref ref-type="disp-formula" rid="e7">Equations 7</xref>, <xref ref-type="disp-formula" rid="e8">8</xref>:<disp-formula id="e7">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mtext>PE</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mtext>in</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mn>1000</mml:mn>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mtext>model</mml:mtext>
</mml:msub>
</mml:mfrac>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mtext>PE</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mn>1000</mml:mn>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mtext>model</mml:mtext>
</mml:msub>
</mml:mfrac>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Architecture of the transformer model.</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g002.tif">
<alt-text content-type="machine-generated">Diagram showing a neural network architecture with separate encoder and decoder components. The encoder includes an embedding layer, position coding, multi-head attention, feedforward neural network (FNN), and add and normalize steps. The decoder features similar elements but also incorporates mask multiple attentions. Both encoder and decoder involve fully connected layers. The process flows from bottom to top, integrating position coding and embedding layers at the base.</alt-text>
</graphic>
</fig>
<p>In the attention mechanism, the input sequence is projected into queries (<inline-formula id="inf22">
<mml:math id="m30">
<mml:mrow>
<mml:mi mathvariant="normal">Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), keys (<inline-formula id="inf23">
<mml:math id="m31">
<mml:mrow>
<mml:mi mathvariant="normal">K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), and values (<inline-formula id="inf24">
<mml:math id="m32">
<mml:mrow>
<mml:mi mathvariant="normal">V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), where <inline-formula id="inf25">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">Q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf26">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">K</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> , and <inline-formula id="inf27">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">V</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are trainable projection matrices. The attention score is computed by taking the dot product of the query and key, followed by scaling and applying the Softmax function, as shown in <xref ref-type="disp-formula" rid="e9">Equation 9</xref>:<disp-formula id="e9">
<mml:math id="m36">
<mml:mrow>
<mml:mtext>Attention</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">V</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Softmax</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">Q</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">K</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:msup>
</mml:mrow>
<mml:msqrt>
<mml:msub>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xb7;</mml:mo>
<mml:mi mathvariant="normal">V</mml:mi>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
<p>Multi-head attention involves parallelizing multiple self-attention heads to capture different feature subspaces. For each head <inline-formula id="inf28">
<mml:math id="m37">
<mml:mrow>
<mml:mi mathvariant="normal">h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the computation can be represented as shown in <xref ref-type="disp-formula" rid="e10">Equation 10</xref>:<disp-formula id="e10">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mtext>ead</mml:mtext>
</mml:mrow>
<mml:mi mathvariant="normal">h</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Attention</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">Q</mml:mi>
<mml:mi mathvariant="normal">h</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">K</mml:mi>
<mml:mi mathvariant="normal">h</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">V</mml:mi>
<mml:mi mathvariant="normal">h</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>The outputs of all heads are concatenated and passed through a linear transformation, as shown in <xref ref-type="disp-formula" rid="e11">Equation 11</xref>:<disp-formula id="e11">
<mml:math id="m39">
<mml:mrow>
<mml:mtext>MultiHead</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">V</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Concat</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mtext>ead</mml:mtext>
</mml:mrow>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mtext>ead</mml:mtext>
</mml:mrow>
<mml:mi mathvariant="normal">h</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">O</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
</p>
<p>The attention output is then passed through a Feed-Forward Network (FFN) consisting of two linear layers with a ReLU activation, as shown in <xref ref-type="disp-formula" rid="e12">Equation 12</xref>:<disp-formula id="e12">
<mml:math id="m40">
<mml:mrow>
<mml:mtext>FFN</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">b</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>Residual connections and Layer Normalization are used after each sub-layer to ensure training stability, as shown in <xref ref-type="disp-formula" rid="e13">Equation 13</xref>:<disp-formula id="e13">
<mml:math id="m41">
<mml:mrow>
<mml:mtext>LayerNorm</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>SubLayer</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
</p>
<p>For time series forecasting, the output of the final Transformer layer is typically the predicted sequence, representing future values at each time step. If <inline-formula id="inf29">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">Z</mml:mi>
<mml:mtext>out</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the output of the final Transformer layer, it can be mapped to the desired prediction dimensions through a linear layer, as shown in <xref ref-type="disp-formula" rid="e14">Equation 14</xref>:<disp-formula id="e14">
<mml:math id="m43">
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">Z</mml:mi>
<mml:mtext>out</mml:mtext>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">y</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">b</mml:mi>
<mml:mi mathvariant="normal">y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-5">
<title>3.5 Overall model workflow and design logic</title>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> presents the overall architecture of the proposed CEEMDAN-GNN-Transformer hybrid framework. The process begins with data collection and preprocessing. The target air quality variable (AQI) is decomposed using CEEMDAN into IMFs and a residual component to reduce non-stationarity and noise.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Overall technical framework of the proposed model.</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g003.tif">
<alt-text content-type="machine-generated">Flowchart showing a data processing system. On the left, data processing involves analysis, collection, and processing. CEEMDAN steps include Original, IMF2, and RES components for signal decomposition. Data is merged and input into a GNN-Transformer system on the right. It includes a GNN model performing spatial relationship mining, followed by a Transformer model generating prediction results. Annotations indicate designing experiments and optimization frameworks.</alt-text>
</graphic>
</fig>
<p>An ablation study is conducted to verify the effectiveness of each model component, and hyperparameters are tuned to achieve optimal performance.</p>
<p>
<statement content-type="step" id="Step_1">
<label>Step 1</label>
<p>The GNN component extracts spatial dependencies among multiple pollutants and decomposed features at each time step, revealing inter-variable relationships.</p>
</statement>
</p>
<p>
<statement content-type="step" id="Step_2">
<label>Step 2</label>
<p>These spatial features are fused with the original features through a designed fusion mechanism, yielding representations suitable for sequence modeling.</p>
</statement>
</p>
<p>
<statement content-type="step" id="Step_3">
<label>Step 3</label>
<p>The Transformer module captures long-term temporal dependencies and nonlinear patterns within the air quality time series, ultimately producing forecasts of future AQI values.</p>
<p>This hybrid framework effectively combines multi-scale signal decomposition, spatial graph modeling, and temporal sequence learning, offering a robust and interpretable approach to air quality forecasting, particularly under the complex conditions present in industrialized urban environments like Chang&#x2019;an Town.</p>
</statement>
</p>
</sec>
</sec>
<sec sec-type="materials" id="s4">
<title>4 Materials</title>
<sec id="s4-1">
<title>4.1 Study area and temporal scope</title>
<p>This study selects Chang&#x2019;an Town in Dongguan City as the research area, as shown in <xref ref-type="fig" rid="F4">Figure 4</xref>, for the empirical analysis of air quality time series forecasting. Chang&#x2019;an Town is one of the most economically developed areas in Dongguan, covering a wide range of industries, including manufacturing, technology, and others. The area has a high concentration of manufacturing enterprises, which results in significant air pollution, thus necessitating precise air quality prediction methods to support environmental management and pollution control.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Geographical location of Chang&#x2019;an town in dongguan city.</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g004.tif">
<alt-text content-type="machine-generated">Map illustrating a city layout with roads, waterways, and labeled highways G425 and G228. A blue marker labeled &#x22;A&#x22; is positioned near the center. Surrounding areas contain urban and green spaces.</alt-text>
</graphic>
</fig>
<p>The dataset spans from 2015 to 2024 and includes 3,647 daily records of major air pollutants: PM<sub>2.5</sub>, PM<sub>10</sub>, NO<sub>2</sub>, SO<sub>2</sub>, CO, and O<sub>3</sub>, alongside the AQI. PM<sub>2.5</sub>, PM<sub>10</sub>, NO<sub>2</sub>, SO<sub>2</sub>, CO, and O<sub>3</sub> are used as predictor variables, while AQI serves as the target variable. The data are sourced from official environmental monitoring platforms with high reliability and accuracy, ensuring the robustness of subsequent modeling tasks.</p>
<p>Situated in the core of the Greater Bay Area, Chang&#x2019;an is surrounded by several industrial cities, and its air quality is influenced by both local emissions and regional pollutant transport, making it highly representative for studying industrial pollution dynamics. The results obtained here are expected to be generalizable to similar urban-industrial zones.</p>
</sec>
<sec id="s4-2">
<title>4.2 Descriptive statistics and feature analysis</title>
<sec id="s4-2-1">
<title>4.2.1 Summary statistics</title>
<p>Descriptive statistics of AQI and major pollutant concentrations are presented in <xref ref-type="table" rid="T1">Table 1</xref>. The data show significant variability, especially in PM<sub>2.5</sub>, PM<sub>10</sub>, and O<sub>3</sub>. The mean AQI is 66, indicating a moderate air quality level. However, its maximum value reaches 207, suggesting occasional severe pollution events. PM<sub>2.5</sub> and PM<sub>10</sub> exhibit high standard deviations, reflecting strong fluctuations in particulate matter concentrations. Although SO<sub>2</sub> and CO concentrations are generally low, CO values are relatively high during certain periods. NO<sub>2</sub> and O<sub>3</sub> also show considerable variability, revealing the complexity of air pollution in the region.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Descriptive statistics of AQI and major pollutants (2015&#x2013;2024).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Metric</th>
<th align="left">AQI</th>
<th align="left">PM<sub>2.5</sub>
</th>
<th align="left">PM<sub>10</sub>
</th>
<th align="left">SO<sub>2</sub>
</th>
<th align="left">CO</th>
<th align="left">NO<sub>2</sub>
</th>
<th align="left">O<sub>3</sub>_8&#xa0;h</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Count</td>
<td align="left">3,647</td>
<td align="left">3,647</td>
<td align="left">3,647</td>
<td align="left">3,647</td>
<td align="left">3,647</td>
<td align="left">3,647</td>
<td align="left">3,647</td>
</tr>
<tr>
<td align="left">Mean</td>
<td align="left">66</td>
<td align="left">46.4</td>
<td align="left">54.24</td>
<td align="left">39.67</td>
<td align="left">4.76</td>
<td align="left">41.01</td>
<td align="left">79.07</td>
</tr>
<tr>
<td align="left">Std Dev</td>
<td align="left">33.26</td>
<td align="left">31.87</td>
<td align="left">31.15</td>
<td align="left">35.35</td>
<td align="left">9.49</td>
<td align="left">35.56</td>
<td align="left">47.62</td>
</tr>
<tr>
<td align="left">Min</td>
<td align="left">17</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">0</td>
<td align="left">0</td>
</tr>
<tr>
<td align="left">25%</td>
<td align="left">41</td>
<td align="left">20</td>
<td align="left">30</td>
<td align="left">9</td>
<td align="left">0.7</td>
<td align="left">10</td>
<td align="left">43</td>
</tr>
<tr>
<td align="left">50%</td>
<td align="left">57</td>
<td align="left">37</td>
<td align="left">49</td>
<td align="left">29</td>
<td align="left">1.1</td>
<td align="left">30</td>
<td align="left">76</td>
</tr>
<tr>
<td align="left">75%</td>
<td align="left">83.5</td>
<td align="left">69</td>
<td align="left">76</td>
<td align="left">65</td>
<td align="left">1.9</td>
<td align="left">68</td>
<td align="left">108</td>
</tr>
<tr>
<td align="left">Max</td>
<td align="left">207</td>
<td align="left">129</td>
<td align="left">175</td>
<td align="left">133</td>
<td align="left">108</td>
<td align="left">120</td>
<td align="left">292</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="fig" rid="F5">Figure 5</xref> illustrates the temporal evolution of AQI and pollutants. Several pollutants, especially SO<sub>2</sub>, show noticeable fluctuations across years, providing a basis for understanding the temporal dynamics and episodic pollution patterns in Chang&#x2019;an.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Time series trends of AQI and major pollutants (2015&#x2013;2024).</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g005.tif">
<alt-text content-type="machine-generated">Line graph depicting air quality indices and pollutants from 2015 to 2025, including AQI, PM2.5, PM10, SO2, CO, NO2, and O3. Each pollutant is represented by a colored line, showing variations over time.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Seasonal variation analysis</title>
<p>To capture seasonal patterns, <xref ref-type="fig" rid="F6">Figure 6</xref> presents the variations of AQI and pollutants across the four seasons: spring, summer, autumn, and winter. The red, blue, and purple curves represent the mean, maximum, and minimum values, respectively.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Seasonal patterns of AQI and major pollutants.</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g006.tif">
<alt-text content-type="machine-generated">Six line graphs show seasonal analysis of air quality indicators: PM2.5, PM10, SO2, CO, NO2, O3_8h, and AQI. Each graph plots mean, max, and min values against seasons: spring, summer, autumn, and winter.</alt-text>
</graphic>
</fig>
<p>Overall, pollutants demonstrate distinct seasonal behaviors. PM<sub>10</sub> and O<sub>3</sub>_8&#xa0;h concentrations peak during summer, possibly due to intense solar radiation enhancing photochemical reactions. In contrast, CO levels rise sharply in winter, likely due to increased emissions from heating and poor atmospheric dispersion. AQI values also tend to be highest in winter, suggesting worse air quality conditions during this season. These patterns highlight the coupled influence of meteorological conditions on pollutant concentrations.</p>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> shows the AQI classification used in China, as officially published by the Ministry of Ecology and Environment. This provides a regulatory context for interpreting the predicted AQI values. This classification serves as a basis for regulatory action and public health advisories, and also provides a meaningful benchmark for evaluating the performance of air quality forecasting models.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>National AQI classification standard (China).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">AQI range</th>
<th align="left">Air quality category</th>
<th colspan="2" align="left">Air quality level</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">0&#x2013;50</td>
<td align="left">Excellent</td>
<td align="left">Level 1</td>
<td align="left" style="background-color:#00E400"/>
</tr>
<tr>
<td align="left">51&#x2013;100</td>
<td align="left">Good</td>
<td align="left">Level 2</td>
<td align="left" style="background-color:#FFFF00"/>
</tr>
<tr>
<td align="left">101&#x2013;150</td>
<td align="left">Lightly Polluted</td>
<td align="left">Level 3</td>
<td align="left" style="background-color:#FF7E00"/>
</tr>
<tr>
<td align="left">151&#x2013;200</td>
<td align="left">Moderately Polluted</td>
<td align="left">Level 4</td>
<td align="left" style="background-color:#FF0000"/>
</tr>
<tr>
<td align="left">201&#x2013;300</td>
<td align="left">Heavily Polluted</td>
<td align="left">Level 5</td>
<td align="left" style="background-color:#99004C"/>
</tr>
<tr>
<td align="left">&#x3e;300</td>
<td align="left">Severely Polluted</td>
<td align="left">Level 6</td>
<td align="left" style="background-color:#7E0023"/>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2-3">
<title>4.2.3 Data distribution</title>
<p>The histograms in <xref ref-type="fig" rid="F7">Figure 7</xref> show that AQI, PM<sub>2.5</sub>, and PM<sub>10</sub> are right-skewed, with a significant peak at low values and a gradual decline afterward, indicating data concentration at the lower end. CO presents a sharp and narrow distribution, reflecting a highly consistent dataset. SO<sub>2</sub>, NO<sub>2</sub> and O<sub>3</sub> exhibit more balanced or slightly left-skewed distributions, implying differences in their formation mechanisms.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Frequency histograms of AQI and major pollutants.</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g007.tif">
<alt-text content-type="machine-generated">Seven histograms display distributions of different air quality indicators: AQI, PM 2.5, PM 10, SO2, CO, NO2, and O3 over eight hours. Each histogram shows frequency distribution with bars and a line indicating the distribution trend. Frequencies vary with each pollutant across different measurement ranges.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4-2-4">
<title>4.2.4 Correlation analysis</title>
<p>
<xref ref-type="fig" rid="F8">Figure 8</xref> illustrates the linear relationships among variables, quantified by Pearson correlation coefficients ranging from &#x2212;1 (strong negative correlation) to &#x2b;1 (strong positive correlation). The color gradient (red for positive, blue for negative) visually emphasizes the strength of each correlation. Notably, PM<sub>2.5</sub> and PM<sub>10</sub> demonstrate a moderately strong positive correlation (approximately 0.30), aligning with expectations for pollutants with shared emission sources and atmospheric dynamics. In contrast, O<sub>3</sub> shows weaker or even negative correlations with other pollutants, indicating a distinct generation mechanism and behavior. Most variable pairs exhibit low or negligible correlations, suggesting that their dynamics are influenced by different factors. These results can guide feature selection and variable engineering in subsequent regression and machine learning analyses.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Pearson correlation coefficient heatmap.</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g008.tif">
<alt-text content-type="machine-generated">Feature correlation heat map displaying correlations among pollutants: AQI, PM2.5, PM10, SO2, CO, NO2, and O3_8h. Positive correlations are shown in red, negative in blue, with a scale from -1.0 to 1.0. Prominent correlations include AQI with itself (1.00), and noteworthy negative correlations like NO2 with SO2 (-0.36).</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="table" rid="T3">Table 3</xref> summarizes the statistical characteristics and normality test results for the AQI data. The AQI ranges from 17 to 207, with a mean of 66.01 and a standard deviation of 33.26, indicating substantial variability. The skewness value of 1.22 suggests a right-skewed distribution, while the kurtosis of 1.40, lower than the theoretical normal value of 3, indicates relatively light tail behavior. The Jarque&#x2013;Bera test yields a statistic of 1,204.26 (p-value &#x3d; 0.00), allowing for the rejection of the normality assumption at a significance level of 0.05. These results confirm that the AQI data deviates significantly from a normal distribution.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Statistical characteristics of AQI and jarque&#x2013;bera (JB) test results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Metric</th>
<th align="left">Value</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Max</td>
<td align="left">207</td>
</tr>
<tr>
<td align="left">Min</td>
<td align="left">17</td>
</tr>
<tr>
<td align="left">Mean</td>
<td align="left">66.01</td>
</tr>
<tr>
<td align="left">Standard Deviation</td>
<td align="left">33.26</td>
</tr>
<tr>
<td align="left">Skewness</td>
<td align="left">1.22</td>
</tr>
<tr>
<td align="left">Kurtosis</td>
<td align="left">1.4</td>
</tr>
<tr>
<td align="left">Jarque-Bera (JB) Test</td>
<td align="left">1,204.26</td>
</tr>
<tr>
<td align="left">p-value</td>
<td align="left">0</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2-5">
<title>4.2.5 Stationarity testing</title>
<p>An Augmented Dickey&#x2013;Fuller (ADF) test was conducted to assess the stationarity of the AQI time series. As shown in <xref ref-type="fig" rid="F9">Figure 9</xref>, the ADF statistic is &#x2212;8.9342, yielding a p-value of 0.0000, well below the significance threshold of 0.05. Thus, the null hypothesis of a unit root (non-stationarity) is rejected, confirming that the AQI series is stationary. This result supports the use of ARIMA models in subsequent baseline forecasting since they require stationary data.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Adf stationarity test results for the AQI time serie.</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g009.tif">
<alt-text content-type="machine-generated">Two graphs display the Autocorrelation Function (ACF) and Partial Autocorrelation Function (PACF) for Air Quality Index (AQI). Both plots show correlation values with lag numbers on the x-axis and correlation coefficients from -1 to 1 on the y-axis. The ACF graph has some points outside the shaded confidence interval, indicating significant autocorrelation, while the PACF shows fewer significant spikes, suggesting limited partial autocorrelation.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s4-3">
<title>4.3 CEEMDAN decomposition of the AQI time series</title>
<p>
<xref ref-type="fig" rid="F10">Figure 10</xref> presents the CEEMDAN decomposition results for the AQI series. CEEMDAN is an improved version of the EMD method. By introducing adaptive noise and performing ensemble averaging, CEEMDAN enhances the stability of the decomposition and improves the separation of modes. The first row (red) shows the original signal, which contains multiscale information, including high-frequency noise and low-frequency trends. The middle green subplots represent different IMFs, arranged from high to low frequencies. The higher-order IMFs capture high-frequency noise and fast oscillations, while the lower-order IMFs gradually reveal low-frequency trends. The final row represents the residual trend component, reflecting the long-term variations of the signal. CEEMDAN provides better decomposition performance than traditional EMD and EEMD (Ensemble Empirical Mode Decomposition), effectively suppressing mode mixing and yielding IMFs with clearer physical significance, making it possible to decompose the target prediction data into interpretable features. With feature engineering completed, the next step involves building the forecasting model and conducting experimental analysis.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>CEEMDAN decomposition of the AQI time series.</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g010.tif">
<alt-text content-type="machine-generated">Graph displaying Intrinsic Mode Functions (IMFs) from an empirical mode decomposition. The top plot shows the original signal in red, followed by ten IMFs in green of varying frequency and amplitude. The bottom plot shows the residual component, labeled &#x22;res.&#x22; Each plot is aligned to a common x-axis spanning from zero to three thousand five hundred.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec sec-type="results|discussion" id="s5">
<title>5 Results and discussion</title>
<sec id="s5-1">
<title>5.1 Experimental environment and data processing</title>
<p>This study implemented the proposed CEEMDAN-GNN-Transformer and baseline models within a Python environment using the PyTorch deep learning framework. All experiments were conducted on a workstation configured with an NVIDIA GeForce RTX 3060 GPU, ensuring consistent training efficiency and reproducibility across experiments. Unless stated otherwise, all experiments were conducted under the same hardware and software conditions.</p>
<p>During the data preprocessing phase, the dataset was divided into training (80%) and test (20%) sets. The feature variables and prediction target were standardized using the StandardScaler method, yielding a mean of 0 and a standard deviation of 1 across both sets. This normalization mitigates the effects of differing variable scales, improves model training efficiency, and promotes stable convergence. The standardized data were then transformed into PyTorch tensors to enable high-performance batch processing within the deep learning pipelines.</p>
</sec>
<sec id="s5-2">
<title>5.2 Model evaluation metrics</title>
<p>To comprehensively assess the performance of the proposed model, this study adopts four widely used regression evaluation metrics: Mean Absolute Error (MAE), Mean Squared Error (MSE), Coefficient of Determination (<italic>R</italic>
<sup>2</sup>), and Explained Variance Score (EVS). These metrics assess model performance from multiple perspectives, including error size, model fit, and the extent of data variance explanation (<xref ref-type="bibr" rid="B5">Chai and Draxler, 2014</xref>; <xref ref-type="bibr" rid="B8">Chicco et al., 2021</xref>).</p>
<p>MAE is the average of the absolute differences between actual observations and predicted values.</p>
<p>MSE is the mean of the squared differences between actual and predicted values.</p>
<p>
<italic>R</italic>
<sup>2</sup> quantifies the model&#x2019;s ability to explain variance in the data, with values closer to 1 indicating better explanatory power.</p>
<p>EVS is similar to <italic>R</italic>
<sup>2</sup> but focuses more on the model&#x2019;s ability to explain data variability, with higher EVS values indicating better fit.</p>
<p>To comprehensively assess the performance of the models in air quality forecasting, four common regression metrics were selected: Mean Absolute Error (MAE), Mean Squared Error (MSE), Coefficient of Determination (<italic>R</italic>
<sup>2</sup>), and Explained Variance Score (EVS). These metrics evaluate the model&#x2019;s prediction performance from multiple dimensions, including error magnitude, goodness of fit, and its ability to capture the data&#x2019;s variability. MAE measures the average absolute difference between the actual observations and the model&#x2019;s predictions, while MSE captures the average squared error between actual and predicted values. The <italic>R</italic>
<sup>2</sup> coefficient quantifies the proportion of data variance explained by the model, ranging from 0 to 1, with values closer to 1 indicating stronger explanatory capacity. EVS operates similarly to <italic>R</italic>
<sup>2</sup> but focuses more explicitly on the proportion of data variance explained by the prediction, making it especially valuable for assessing the model&#x2019;s ability to capture variability across the dataset.</p>
</sec>
<sec id="s5-3">
<title>5.3 Baseline models</title>
<p>To validate the effectiveness of the proposed CEEMDAN-GNN-Transformer model, this study selects nine representative predictive models for comparison. These models include deep learning approaches (e.g., CNN-Transformer), ensemble learning methods (e.g., XGBoost, GBDT, Random Forest), traditional machine learning methods (e.g., KNN, SVM), and classical time series models (e.g., ARIMA). These models exhibit significant differences in terms of time pattern modeling ability, variable relationship modeling strength, IMF component utilization, implementation complexity, and computational resource requirements, as summarized in <xref ref-type="table" rid="T4">Table 4</xref> (<xref ref-type="bibr" rid="B9">Cui et al., 2023</xref>; <xref ref-type="bibr" rid="B28">Ye et al., 2025</xref>; <xref ref-type="bibr" rid="B17">Lei et al., 2023</xref>).</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>GNN-transformer and baseline model feature comparison.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Model</th>
<th align="left">Time pattern</th>
<th align="left">Variable relationship</th>
<th align="left">IMF component utilization</th>
<th align="left">Implementation complexity</th>
<th align="left">Scale requirements</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">GNN-Transformer</td>
<td align="left">Medium (depends on Transformer)</td>
<td align="left">Strong (explicit modeling)</td>
<td align="left">Strong (hierarchical modeling)</td>
<td align="left">Medium-high</td>
<td align="left">Medium-high</td>
</tr>
<tr>
<td align="left">CNN-Transformer</td>
<td align="left">Strong (local &#x2b; global)</td>
<td align="left">Weak (implicit modeling)</td>
<td align="left">Medium (time scale decomposition)</td>
<td align="left">Medium-high</td>
<td align="left">Low</td>
</tr>
<tr>
<td align="left">XGBoost</td>
<td align="left">Medium (depends on windowing)</td>
<td align="left">Medium (feature interaction)</td>
<td align="left">Medium (used as features)</td>
<td align="left">Medium</td>
<td align="left">Medium</td>
</tr>
<tr>
<td align="left">AdaBoost</td>
<td align="left">Medium (depends on windowing)</td>
<td align="left">Medium (feature interaction)</td>
<td align="left">Medium (used as features)</td>
<td align="left">Medium</td>
<td align="left">Medium</td>
</tr>
<tr>
<td align="left">KNN</td>
<td align="left">Weak (no explicit modeling)</td>
<td align="left">Weak (no explicit modeling)</td>
<td align="left">Weak (no hierarchical awareness)</td>
<td align="left">Low</td>
<td align="left">Low</td>
</tr>
<tr>
<td align="left">GBDT</td>
<td align="left">Medium (depends on windowing)</td>
<td align="left">Medium (feature interaction)</td>
<td align="left">Medium (used as features)</td>
<td align="left">Medium</td>
<td align="left">Medium</td>
</tr>
<tr>
<td align="left">SVM</td>
<td align="left">Medium (depends on windowing)</td>
<td align="left">Medium (depends on kernel)</td>
<td align="left">Medium (depends on feature engineering)</td>
<td align="left">Medium-high (kernel parameters)</td>
<td align="left">Medium-high</td>
</tr>
<tr>
<td align="left">Random Forest</td>
<td align="left">Medium (depends on windowing)</td>
<td align="left">Medium (feature interaction)</td>
<td align="left">Medium (used as features)</td>
<td align="left">Medium</td>
<td align="left">Medium</td>
</tr>
<tr>
<td align="left">ARIMA</td>
<td align="left">Strong (explicit time pattern modeling)</td>
<td align="left">Weak (univariate)</td>
<td align="left">Weak (low utilization)</td>
<td align="left">Low</td>
<td align="left">Low</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Through comparative analysis, it is evident that CNN-Transformer combines CNN&#x2019;s local feature extraction ability and Transformer&#x2019;s global dependency modeling capability, making it highly suitable for time series modeling. Compared to traditional time series models, CNN-Transformer efficiently extracts short-term patterns while modeling long-range dependencies, with relatively low computational complexity. By comparing it with GNN-Transformer, this study highlights the necessity of the GNN structure for temporal tasks and examines whether it enhances the Transformer&#x2019;s performance in complex time series modeling. The comparison validates the advantage and applicability of GNN-Transformer in handling complex time series data and relational modeling tasks.</p>
</sec>
<sec id="s5-4">
<title>5.4 Ablation study</title>
<p>To explore the contribution of each key module in the proposed model and verify the combination effect, this study designs a systematic ablation experiment. Keeping hyperparameters consistent, the experiment sequentially removes or combines the signal decomposition module (CEEMDAN), the GNN, and the Transformer module, with a Multi-Layer Perceptron (MLP) used as the baseline model for comparison. Each experiment is repeated 20 times, with the average result taken to ensure robust and fair evaluation. The results are shown in <xref ref-type="table" rid="T5">Table 5</xref>.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Ablation experiment results comparison.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Component</th>
<th align="left">MAE</th>
<th align="left">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">MLP (Baseline)</td>
<td align="left">16.97</td>
<td align="left">0.43</td>
</tr>
<tr>
<td align="left">GNN</td>
<td align="left">15.422</td>
<td align="left">0.48</td>
</tr>
<tr>
<td align="left">Transformer</td>
<td align="left">14.032</td>
<td align="left">0.54</td>
</tr>
<tr>
<td align="left">GNN-Transformer</td>
<td align="left">10.021</td>
<td align="left">0.79</td>
</tr>
<tr>
<td align="left">CEEMDAN-GNN</td>
<td align="left">14.221</td>
<td align="left">0.58</td>
</tr>
<tr>
<td align="left">CEEMDAN-Transformer</td>
<td align="left">8.9022</td>
<td align="left">0.81</td>
</tr>
<tr>
<td align="left">CEEMDAN-GNN-Transformer</td>
<td align="left">9.2495</td>
<td align="left">0.87</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The experimental results indicate that both GNN and Transformer significantly improve the model&#x2019;s performance. Transformer, in particular, demonstrates a more substantial effect (MAE reduced from 16.97 to 14.032, <italic>R</italic>
<sup>2</sup> improved to 0.54). The combination of GNN and Transformer further enhances prediction capability (MAE reduced to 10.021, <italic>R</italic>
<sup>2</sup> improved to 0.79), indicating a synergistic effect. Moreover, the introduction of CEEMDAN signal decomposition significantly optimizes the Transformer&#x2019;s prediction performance (MAE reduced to 8.9022, <italic>R</italic>
<sup>2</sup> improved to 0.81). When CEEMDAN is added on top of the GNN-Transformer combination, the performance improvement is limited (MAE decreases from 10.021 to 9.2495, <italic>R</italic>
<sup>2</sup> improves to 0.87). In summary, the CEEMDAN-GNN-Transformer architecture demonstrates superior modeling capability in time series prediction tasks, with the complementary strengths of the three components achieving the best performance in complex temporal modeling tasks.</p>
</sec>
<sec id="s5-5">
<title>5.5 Hyperparameter optimization and model training</title>
<p>This study employs the Optuna framework for hyperparameter optimization, aiming to minimize the MSE of the GNN and Transformer models on the validation set. Optuna is a Bayesian optimization-based automated hyperparameter search tool that intelligently explores the search space to find the optimal combination of hyperparameters with the fewest trials. It supports various machine learning and deep learning frameworks, including Scikit-learn, PyTorch, and TensorFlow. By defining the search space, the optimizer selects appropriate hyperparameters such as the number of attention heads, hidden layer dimensions, model depth, learning rate, dropout rate, and number of neighbors. Through 50 trials, Optuna automatically tunes the hyperparameters to find the best combination, improving model prediction accuracy, reducing computational cost, and enhancing generalization ability.</p>
<p>
<xref ref-type="table" rid="T6">Table 6</xref> lists various hyperparameters affecting model performance, along with their respective search ranges. These include structural parameters (scaling factor for hidden layers, attention heads, and encoder layers), graph structure parameters (number of neighboring nodes in the graph), regularization parameters (dropout rate), and optimizer parameters (learning rate). The tuning of these hyperparameters directly impacts model training performance and generalization ability. After 20 iterations, the optimal combination of hyperparameters was obtained. A partial example of the optimization results is shown in <xref ref-type="table" rid="T7">Table 7</xref>.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Hyperparameter search space.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Category</th>
<th align="left">Hyperparameter name</th>
<th align="left">Search range</th>
<th align="left">Affected model component</th>
<th align="left">Description</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Scaling Coefficients</td>
<td align="left">hidden_dim_factor</td>
<td align="left">16&#x2013;64</td>
<td align="left">GNN - Transformer</td>
<td align="left">Controls the scaling factor for hidden layers</td>
</tr>
<tr>
<td align="left">Structural Parameters</td>
<td align="left">num_heads</td>
<td align="left">2&#x2013;8</td>
<td align="left">Transformer</td>
<td align="left">Number of attention heads</td>
</tr>
<tr>
<td align="left">Structural Parameters</td>
<td align="left">num_layers</td>
<td align="left">1&#x2013;3</td>
<td align="left">Transformer</td>
<td align="left">Number of encoder layers</td>
</tr>
<tr>
<td align="left">Graph Structure Parameters</td>
<td align="left">k_neighbors</td>
<td align="left">3&#x2013;10</td>
<td align="left">GNN</td>
<td align="left">Number of neighboring nodes in the graph</td>
</tr>
<tr>
<td align="left">Regularization Parameters</td>
<td align="left">dropout</td>
<td align="left">0.1&#x2013;0.5</td>
<td align="left">GNN - Transformer</td>
<td align="left">Dropout rate</td>
</tr>
<tr>
<td align="left">Optimizer Parameters</td>
<td align="left">lr</td>
<td align="left">0.0001&#x2013;0.01</td>
<td align="left">Overall Model</td>
<td align="left">Learning rate</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Hyperparameter optimization results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Iteration</th>
<th align="left">Loss</th>
<th align="left">num_heads</th>
<th align="left">hidden_dim_factor</th>
<th align="left">num_layers</th>
<th align="left">Lr</th>
<th align="left">Dropout</th>
<th align="left">k_neighbors</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">0</td>
<td align="left">0.463</td>
<td align="left">3</td>
<td align="left">44</td>
<td align="left">3</td>
<td align="left">0.0001</td>
<td align="left">0.18658</td>
<td align="left">9</td>
</tr>
<tr>
<td align="left">1</td>
<td align="left">0.283</td>
<td align="left">7</td>
<td align="left">34</td>
<td align="left">3</td>
<td align="left">0.0001</td>
<td align="left">0.13044</td>
<td align="left">7</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">0.27</td>
<td align="left">5</td>
<td align="left">60</td>
<td align="left">1</td>
<td align="left">0.0002</td>
<td align="left">0.19506</td>
<td align="left">6</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">0.27</td>
<td align="left">8</td>
<td align="left">19</td>
<td align="left">2</td>
<td align="left">0.0004</td>
<td align="left">0.3713</td>
<td align="left">6</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">0.51</td>
<td align="left">2</td>
<td align="left">59</td>
<td align="left">2</td>
<td align="left">0.0016</td>
<td align="left">0.4443</td>
<td align="left">7</td>
</tr>
<tr>
<td align="left">
<bold>5</bold>
</td>
<td align="left">
<bold>0.192</bold>
</td>
<td align="left">
<bold>8</bold>
</td>
<td align="left">
<bold>47</bold>
</td>
<td align="left">
<bold>1</bold>
</td>
<td align="left">
<bold>0.00015</bold>
</td>
<td align="left">
<bold>0.1969</bold>
</td>
<td align="left">
<bold>5</bold>
</td>
</tr>
<tr>
<td align="left">&#x2026;</td>
<td align="left">&#x2026;</td>
<td align="left">&#x2026;</td>
<td align="left">&#x2026;</td>
<td align="left">&#x2026;</td>
<td align="left">&#x2026;</td>
<td align="left">&#x2026;</td>
<td align="left">&#x2026;</td>
</tr>
<tr>
<td align="left">46</td>
<td align="left">0.255</td>
<td align="left">7</td>
<td align="left">28</td>
<td align="left">2</td>
<td align="left">0.0007</td>
<td align="left">0.35967</td>
<td align="left">8</td>
</tr>
<tr>
<td align="left">47</td>
<td align="left">0.766</td>
<td align="left">6</td>
<td align="left">56</td>
<td align="left">3</td>
<td align="left">0.0008</td>
<td align="left">0.273</td>
<td align="left">9</td>
</tr>
<tr>
<td align="left">48</td>
<td align="left">0.509</td>
<td align="left">7</td>
<td align="left">35</td>
<td align="left">2</td>
<td align="left">0.0003</td>
<td align="left">0.27438</td>
<td align="left">7</td>
</tr>
<tr>
<td align="left">49</td>
<td align="left">0.676</td>
<td align="left">8</td>
<td align="left">45</td>
<td align="left">1</td>
<td align="left">0.0006</td>
<td align="left">0.33695</td>
<td align="left">5</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>
<sup>&#x2a;</sup>The bolded row indicates the best hyperparameter combination obtained in this optimization.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="T7">Table 7</xref>, the optimal hyperparameter combination found through the optimization process includes 8 attention heads, a hidden dimension scaling factor of 47, 1 attention layer, a learning rate of 0.000153, a dropout rate of 0.197, and 5 neighbors. This combination resulted in the best performance, with a loss of 0.1929, demonstrating the model&#x2019;s best performance under these settings. The actual hidden layer dimension is calculated by multiplying the hidden_dim_factor by num_heads.</p>
<p>The final model was built using the best parameters obtained from Optuna&#x2019;s optimization. The following steps were undertaken: the graph structure was constructed, with adjacency matrices calculated based on feature similarities for both the training and testing sets, where the k nearest neighbors were identified for each data point. The edge index for GNN computation was then built. The same k_neighbors value was used for both training and testing sets, but separate graph structures were created for each. The model was trained for 200 epochs. Each epoch included a forward pass through the GNN layers, Transformer layers, and a fully connected layer, followed by calculating the mean squared error loss between predicted and true values. Backpropagation was performed to compute gradients, and the Adam optimizer was used to update the model parameters.</p>
<p>From the training logs, as shown in <xref ref-type="table" rid="T8">Table 8</xref>, the loss decreases progressively over the training process, indicating that the model is converging and improving its prediction accuracy. The loss decreases rapidly during the initial stages (Epochs 20&#x2013;40), from 0.3592 to 0.2892, reflecting the larger optimization steps in the early training phase. In subsequent stages, the loss stabilizes, indicating that the training process is becoming more stable. By Epoch 200, the loss converges to 0.1635. Overall, the decreasing trend in loss indicates the model&#x2019;s effective training and suggests that it achieves an optimal performance level.</p>
<table-wrap id="T8" position="float">
<label>TABLE 8</label>
<caption>
<p>Model training loss progression.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Epoch</th>
<th align="left">20</th>
<th align="left">40</th>
<th align="left">60</th>
<th align="left">80</th>
<th align="left">100</th>
<th align="left">120</th>
<th align="left">140</th>
<th align="left">160</th>
<th align="left">180</th>
<th align="left">200</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Loss</td>
<td align="left">0.3592</td>
<td align="left">0.2892</td>
<td align="left">0.2403</td>
<td align="left">0.2187</td>
<td align="left">0.2027</td>
<td align="left">0.1866</td>
<td align="left">0.1748</td>
<td align="left">0.1662</td>
<td align="left">0.1631</td>
<td align="left">0.1635</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="table" rid="T9">Table 9</xref> provides a summary of the final model&#x2019;s parameter counts, highlighting its complexity and capacity.</p>
<table-wrap id="T9" position="float">
<label>TABLE 9</label>
<caption>
<p>Final model architecture and parameter counts.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Layer name</th>
<th align="left">Parameter shape</th>
<th align="left">Parameter count</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">gnn1.bias</td>
<td align="left">(376)</td>
<td align="left">376</td>
</tr>
<tr>
<td align="left">gnn1.lin.weight</td>
<td align="left">(376, 17)</td>
<td align="left">6,392</td>
</tr>
<tr>
<td align="left">gnn2.bias</td>
<td align="left">(376)</td>
<td align="left">376</td>
</tr>
<tr>
<td align="left">gnn2.lin.weight</td>
<td align="left">(376, 376)</td>
<td align="left">141,376</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.self_attn.in_proj_weight</td>
<td align="left">(1,128, 376)</td>
<td align="left">424,128</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.self_attn.in_proj_bias</td>
<td align="left">(1,128)</td>
<td align="left">1,128</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.self_attn.out_proj.weight</td>
<td align="left">(376, 376)</td>
<td align="left">141,376</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.self_attn.out_proj.bias</td>
<td align="left">(376)</td>
<td align="left">376</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.linear1.weight</td>
<td align="left">(1,504, 376)</td>
<td align="left">565,504</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.linear1.bias</td>
<td align="left">(1,504)</td>
<td align="left">1,504</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.linear2.weight</td>
<td align="left">(376, 1,504)</td>
<td align="left">565,504</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.linear2.bias</td>
<td align="left">(376)</td>
<td align="left">376</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.norm1.weight</td>
<td align="left">(376)</td>
<td align="left">376</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.norm1.bias</td>
<td align="left">(376)</td>
<td align="left">376</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.norm2.weight</td>
<td align="left">(376)</td>
<td align="left">376</td>
</tr>
<tr>
<td align="left">transformer_encoder.layers.0.norm2.bias</td>
<td align="left">(376)</td>
<td align="left">376</td>
</tr>
<tr>
<td align="left">fc.weight</td>
<td align="left">(1, 376)</td>
<td align="left">376</td>
</tr>
<tr>
<td align="left">fc.bias</td>
<td align="left">(1)</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">
<bold>Total Parameters</bold>
</td>
<td align="left"/>
<td align="left">
<bold>1,850,297</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>
<sup>&#x2a;</sup>The bolded row indicates the total number of trainable parameters in the model.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The model developed in this study integrates a GNN and a Transformer architecture to process graph-structured data and capture global dependencies. The input feature dimension is 17, and after two layers of GNN processing, it is mapped to a 376-dimensional hidden representation. Subsequently, a single-layer Transformer encoder with 8 attention heads is used to further model global feature interactions. Finally, a fully connected layer maps the features to a 1-dimensional output for pollutant prediction. The total number of parameters is approximately 1.85 million, making it a relatively complex model for pollutant time series forecasting.</p>
</sec>
<sec id="s5-6">
<title>5.6 Forecasting results and comparative analysis</title>
<p>To comprehensively evaluate the forecasting performance of the proposed CEEMDAN&#x2013;GNN&#x2013;Transformer model for air quality prediction, a comparative analysis was conducted against a range of classical machine learning and deep learning approaches. All models were tested on the same dataset preprocessed via CEEMDAN signal decomposition, and their performance was assessed using four commonly used metrics: MAE, MSE, <italic>R</italic>
<sup>2</sup>, and EVS. The results are presented in <xref ref-type="table" rid="T10">Table 10</xref>.</p>
<table-wrap id="T10" position="float">
<label>TABLE 10</label>
<caption>
<p>Model performance comparison (based on CEEMDAN decomposition).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Model</th>
<th align="left">MAE</th>
<th align="left">MSE</th>
<th align="left">
<italic>R</italic>
<sup>2</sup>
</th>
<th align="left">EVS</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">XGBoost</td>
<td align="left">8.1333</td>
<td align="left">103.2361</td>
<td align="left">0.889</td>
<td align="left">0.9044</td>
</tr>
<tr>
<td align="left">Random Forest</td>
<td align="left">9.373</td>
<td align="left">144.0375</td>
<td align="left">0.8451</td>
<td align="left">0.8628</td>
</tr>
<tr>
<td align="left">Adaboost</td>
<td align="left">14.2955</td>
<td align="left">283.4626</td>
<td align="left">0.6952</td>
<td align="left">0.788</td>
</tr>
<tr>
<td align="left">KNN</td>
<td align="left">14.8311</td>
<td align="left">422.7499</td>
<td align="left">0.5454</td>
<td align="left">0.5597</td>
</tr>
<tr>
<td align="left">SVM</td>
<td align="left">10.4308</td>
<td align="left">129.8963</td>
<td align="left">0.8603</td>
<td align="left">0.9567</td>
</tr>
<tr>
<td align="left">ARIMA</td>
<td align="left">17.3121</td>
<td align="left">354.4236</td>
<td align="left">0.4989</td>
<td align="left">0.4835</td>
</tr>
<tr>
<td align="left">CNN-Transformer</td>
<td align="left">10.6673</td>
<td align="left">168.2338</td>
<td align="left">0.8173</td>
<td align="left">0.8247</td>
</tr>
<tr>
<td align="left">GNN-Transformer</td>
<td align="left">7.6495</td>
<td align="left">98.2009</td>
<td align="left">0.9192</td>
<td align="left">0.9491</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="T10">Table 10</xref>, the forecasting performances of the compared models vary significantly. The GNN&#x2013;Transformer approach achieves the best results across all metrics, with the lowest MAE (7.6495) and MSE (98.2009), along with the highest <italic>R</italic>
<sup>2</sup> (0.9192) and EVS (0.9491). This highlights its superior ability to capture complex temporal dynamics and to accurately explain the variance within the data. Meanwhile, XGBoost also delivers strong performance, yielding an <italic>R</italic>
<sup>2</sup> of 0.8890 and an EVS of 0.9044, underscoring its effectiveness in capturing feature interactions and nonlinear relationships. In comparison, Random Forest, SVM, and CNN-Transformer perform slightly worse than XGBoost and GNN-Transformer in terms of MSE and MAE, while Adaboost and KNN show relatively weaker performance in all metrics. Notably, KNN has the highest MSE (422.7499) and the lowest <italic>R</italic>
<sup>2</sup> (0.5454), which indicates its limitations in modeling complex dynamic relationships, especially in high-dimensional regression tasks. Overall, the GNN-Transformer and XGBoost models demonstrate superior prediction performance in this study, while traditional machine learning methods (e.g., KNN, Adaboost) show certain limitations in modeling complex dynamic changes.</p>
<p>
<xref ref-type="fig" rid="F11">Figure 11</xref> illustrates the forecasting results of the GNN-Transformer model compared with observed AQI values. The predicted time series closely tracks the actual observations, demonstrating the model&#x2019;s strong temporal modeling capability and its ability to capture long-term trends in air quality fluctuations. However, slight discrepancies are observed during certain extreme events, such as pollution peaks or sharp declines, where the predicted values tend to be smoother than the actual data. This limitation may arise from the signal decomposition and denoising process of CEEMDAN, which can attenuate high-frequency variations. Despite this, the proposed GNN-Transformer approach shows considerable promise in capturing the overall dynamics of air quality. In future work, incorporating meteorological covariates or further refining the model architecture could help better capture these abrupt variations and further improve prediction precision.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>GNN-transformer model prediction fitting.</p>
</caption>
<graphic xlink:href="fenvs-13-1653446-g011.tif">
<alt-text content-type="machine-generated">Line graph comparing true and predicted values. The x-axis shows index numbers from 0 to 750, and the y-axis measures values from 0 to 200. Blue lines represent true values, while orange lines show predicted values. The graph title reads &#x22;Comparison of the final model prediction results.&#x22;</alt-text>
</graphic>
</fig>
<p>Although CEEMDAN is effective in reducing noise and extracting multi-scale features, its decomposition inherently introduces a smoothing bias, diminishing the retention of high-frequency signals. In this study, extreme pollution episodes (such as sharp AQI surges or declines) appear attenuated in the fitted curve. While beneficial for overall stability and trend extraction, this effect may be suboptimal for real-time air quality management, where timely detection of extreme events is critical for public health and emergency response. Future work could consider augmenting CEEMDAN with high-frequency component weighting or integrating auxiliary submodels optimized for peak detection to enhance responsiveness to short-term pollution shocks.</p>
</sec>
<sec id="s5-7">
<title>5.7 Model uncertainty quantification and prediction interval evaluation</title>
<p>In environmental air quality forecasting, beyond accuracy, it is equally important to quantify the uncertainty of predictions, as this information supports scientific decision-making and risk management. To this end, the Monte Carlo (MC) Dropout method&#x2014;a Bayesian approximation technique&#x2014;was applied to the CEEMDAN&#x2013;GNN&#x2013;Transformer model for uncertainty analysis.</p>
<p>Originally proposed by <xref ref-type="bibr" rid="B10">Gal and Ghahramani. (2016)</xref>, MC Dropout involves keeping the dropout mechanism active during inference, performing multiple stochastic forward passes to simulate sampling from the posterior weight distribution. This enables approximate estimation of the predictive distribution. For each of T stochastic passes, predictions <inline-formula id="inf30">
<mml:math id="m44">
<mml:mrow>
<mml:mover accent="true">
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:msub>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> are collected, from which the predictive mean and variance are computed, as shown in <xref ref-type="disp-formula" rid="e15">Equations 15</xref>, <xref ref-type="disp-formula" rid="e16">16</xref>:<disp-formula id="e15">
<mml:math id="m45">
<mml:mrow>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:msup>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2248;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mover accent="true">
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:msub>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
<disp-formula id="e16">
<mml:math id="m46">
<mml:mrow>
<mml:mtext>Var</mml:mtext>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:msup>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2248;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
</mml:msub>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:msup>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
</p>
<p>Using these results, two uncertainty evaluation metrics were calculated: Prediction Interval Coverage Probability (PICP) and Mean Prediction Interval Width (MPIW). The results are shown in <xref ref-type="table" rid="T11">Table 11</xref>.</p>
<table-wrap id="T11" position="float">
<label>TABLE 11</label>
<caption>
<p>Bayesian approximation&#x2013;based uncertainty evaluation.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Metric</th>
<th align="left">Value</th>
<th align="left">Ideal/Target</th>
<th align="left">Interpretation</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Point prediction accuracy</td>
<td align="left">0.902</td>
<td align="left">&#x2192; 1</td>
<td align="left">High point-prediction performance</td>
</tr>
<tr>
<td align="left">PICP</td>
<td align="left">0.916</td>
<td align="left">&#x2265;0.95</td>
<td align="left">Interval covers most true values</td>
</tr>
<tr>
<td align="left">MPIW</td>
<td align="left">22.001</td>
<td align="left">Context-dependent</td>
<td align="left">Narrow interval, high precision</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The results show that the PICP (0.916) is close to the ideal 95% coverage, indicating that the prediction intervals successfully cover the majority of observed values. The MPIW (22.001) is substantially narrower than the typical AQI category width (&#x2248;50 for mild pollution), implying high-precision prediction intervals.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>This study addresses the forecasting of the AQI in Changan Town, Dongguan, and proposes an improved CEEMDAN&#x2013;GNN&#x2013;Transformer framework aimed at enhancing both prediction accuracy and stability. First, the AQI time series is decomposed using the CEEMDAN method, extracting multi-scale features and mitigating the effects of non-stationarity in the original data. Second, a GNN is introduced to capture spatial dependencies, effectively uncovering latent spatial relationships among various pollutants and environmental factors, thereby enhancing the model&#x2019;s ability to perceive the evolution patterns of AQI. Finally, a Transformer module is employed to model long-term temporal dependencies, capturing the temporal dynamics and trend characteristics of AQI variations.</p>
<p>Through ablation experiments and hyperparameter optimization, this study validated the contribution of each submodule to the overall model performance and compared it with several mainstream baseline models. The experimental results show that the proposed CEEMDAN-GNN-Transformer model outperforms all other models across all evaluation metrics (MAE, MSE, <italic>R</italic>
<sup>2</sup>, EVS), with particularly significant advantages in prediction accuracy and fitting ability. Specifically, the model achieved the lowest MAE (7.6495) and MSE (98.2009), an <italic>R</italic>
<sup>2</sup> of 0.9192, and an EVS of 0.9491, demonstrating that this approach can accurately capture the long-term evolution characteristics of AQI and improve the reliability and stability of the prediction results.</p>
<p>From a deployment perspective, the CEEMDAN&#x2013;GNN&#x2013;Transformer model demonstrates manageable hardware requirements. Experiments indicate that training and inference can be completed within a reasonable time frame on a mid-range GPU (NVIDIA RTX 3060), making it viable for implementation in local environmental monitoring centers or regional environmental agencies. However, several limitations remain: (1) current inputs are limited to pollutant concentration data, without incorporating meteorological conditions or emission source activity, which may reduce accuracy during extreme pollution events; (2) the smoothing effect of CEEMDAN, while improving stability, may weaken sensitivity to short-term abrupt changes; and (3) cross-regional generalization has yet to be validated with multi-city, multi-station datasets.</p>
<p>Future work will focus on integrating multi-source data (e.g., meteorological variables, traffic flow) and introducing weighted enhancement strategies for high-frequency components to balance long-term trend forecasting with short-term anomaly detection. Furthermore, interpretability methods will be incorporated&#x2014;leveraging attention weight distributions and feature attribution rankings&#x2014;to systematically analyze the relative contributions of pollutants and environmental factors, thereby improving transparency and application value. Finally, the model will be tested across diverse regional settings to promote its practical adoption in real-time AQI early warning and long-term trend prediction.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s13">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>SH: Conceptualization, Methodology, Writing &#x2013; original draft, Writing &#x2013; review and editing, Funding acquisition. YL: Investigation, Data curation, Writing &#x2013; review and editing. JP: Formal analysis, Data curation, Writing &#x2013; review and editing. DX: Supervision, Resources, Writing &#x2013; review and editing. YW: Methodology, Data curation, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was supported by the following projects: the Doctoral Research Start-up Fund Project of Guangdong University of Science and Technology (Grant No. GKY-2024BSQDW-6), the Project of China Business Statistics Society (Grant No. 2024STY118), the Guangdong Social Science Fund Special Project &#x201c;Mechanism of Original Innovation in Emerging Leading Industries in the Guangdong-Hong Kong-Macao Greater Bay Area&#x201d; (Grant No. GD24DWQGL01), the Guangdong Basic and Applied Basic Research Joint Fund Project &#x201c;The Role of the Guangdong-Dongguan Joint Fund in Promoting Regional Original Innovation&#x201d; (Grant No. 2023A1515140047), the Guangdong Philosophy and Social Science Planning Co-construction Project &#x201c;Policy Perception and Simulation of the Digital Transformation of Manufacturing in Guangdong Province&#x201d; (Grant No. GD23XGL052), and the Dongguan Philosophy and Social Science Planning General Project &#x201c;Development Path and Practical Strategies of Manufacturing Aesthetics in Dongguan from the Perspective of the Digital Economy Empowerment&#x201d; (Grant No. 2025CG105).</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s13">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fenvs.2025.1653446/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fenvs.2025.1653446/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.zip" id="SM1" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abirami</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chitra</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Regional air quality forecasting using spatiotemporal deep learning</article-title>. <source>J. Clean. Prod.</source> <volume>283</volume>, <fpage>125341</fpage>. <pub-id pub-id-type="doi">10.1016/j.jclepro.2020.125341</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Agbehadji</surname>
<given-names>I. E.</given-names>
</name>
<name>
<surname>Obagbuwa</surname>
<given-names>I. C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Systematic review of machine learning and deep learning techniques for spatiotemporal air quality prediction</article-title>. <source>Atmosphere</source> <volume>15</volume> (<issue>11</issue>), <fpage>1352</fpage>. <pub-id pub-id-type="doi">10.3390/atmos15111352</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ameri</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Band</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Zamani</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shu</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Khorsandroo</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Forecasting PM 2.5 concentration based on integrating of CEEMDAN decomposition method with SVM and LSTM</article-title>. <source>Ecotoxicol. Environ. Saf.</source> <volume>266</volume>, <fpage>115572</fpage>. <pub-id pub-id-type="doi">10.1016/j.ecoenv.2023.115572</pub-id>
<pub-id pub-id-type="pmid">37837695</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ban</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>PM2. 5 prediction based on the CEEMDAN algorithm and a machine learning hybrid model</article-title>. <source>Sustainability</source> <volume>14</volume> (<issue>23</issue>), <fpage>16128</fpage>. <pub-id pub-id-type="doi">10.3390/su142316128</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chai</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Draxler</surname>
<given-names>R. R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Root mean square error (RMSE) or mean absolute error (MAE)? Arguments against avoiding RMSE in the literature</article-title>. <source>Geosci. Model Dev.</source> <volume>7</volume> (<issue>3</issue>), <fpage>1247</fpage>&#x2013;<lpage>1250</lpage>. <pub-id pub-id-type="doi">10.5194/gmd-7-1247-2014</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A hybrid CNN-Transformer model for ozone concentration prediction</article-title>. <source>Air Qual. Atmos. and Health</source> <volume>15</volume> (<issue>9</issue>), <fpage>1533</fpage>&#x2013;<lpage>1546</lpage>. <pub-id pub-id-type="doi">10.1007/s11869-022-01197-w</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Group-aware graph neural network for nationwide city air quality forecasting</article-title>. <source>ACM Trans. Knowl. Discov. Data</source> <volume>18</volume> (<issue>3</issue>), <fpage>1</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1145/3631713</pub-id>
<pub-id pub-id-type="pmid">40264574</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chicco</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Warrens</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Jurman</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The coefficient of determination R-squared is more informative than SMAPE, MAE, MAPE, MSE and RMSE in regression analysis evaluation</article-title>. <source>Peerj Comput. Sci.</source> <volume>7</volume>, <fpage>e623</fpage>. <pub-id pub-id-type="doi">10.7717/peerj-cs.623</pub-id>
<pub-id pub-id-type="pmid">34307865</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cui</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Deep learning methods for atmospheric PM2. 5 prediction: a comparative study of transformer and CNN-LSTM-attention</article-title>. <source>Atmos. Pollut. Res.</source> <volume>14</volume> (<issue>9</issue>), <fpage>101833</fpage>. <pub-id pub-id-type="doi">10.1016/j.apr.2023.101833</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gal</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ghahramani</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Dropout as a bayesian approximation: representing model uncertainty in deep learning</article-title>. In: <source>International conference on machine learning</source>. <publisher-name>Dongguan, China: University of Science and Technology</publisher-name>. p. <fpage>1050</fpage>&#x2013;<lpage>1059</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ge</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Multi-scale spatiotemporal graph convolution network for air quality prediction</article-title>. <source>Appl. Intell.</source> <volume>51</volume>, <fpage>3491</fpage>&#x2013;<lpage>3505</lpage>. <pub-id pub-id-type="doi">10.1007/s10489-020-02054-y</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<collab>Guangdong Provincial Department of Ecology and Environment</collab> (<year>2023</year>). <source>Guangdong ecological and environmental bulletin 2022 [in Chinese]</source>. <publisher-name>Guangzhou: Guangdong Provincial Department of Ecology and Environment</publisher-name>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A new time series forecasting model based on complete ensemble empirical mode decomposition with adaptive noise and temporal convolutional network</article-title>. <source>Neural Process. Lett.</source> <volume>55</volume> (<issue>4</issue>), <fpage>4397</fpage>&#x2013;<lpage>4417</lpage>. <pub-id pub-id-type="doi">10.1007/s11063-022-11046-7</pub-id>
<pub-id pub-id-type="pmid">36248248</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Hybrid transformer-TimesNet model for accurate prediction of industrial air pollutants: a case study of the xinyang industrial zone, China</article-title>. <source>Pol. J. Environ. Stud.</source> <volume>152</volume>, <fpage>203348</fpage>. <pub-id pub-id-type="doi">10.15244/pjoes/203348</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Horn</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Dasgupta</surname>
<given-names>P. K.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>The Air Quality Index (AQI) in historical and analytical perspective a tutorial review</article-title>. <source>Talanta</source> <volume>267</volume>, <fpage>125260</fpage>. <pub-id pub-id-type="doi">10.1016/j.talanta.2023.125260</pub-id>
<pub-id pub-id-type="pmid">37852126</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Adaptive denoising algorithm using peak statistics-based thresholding and novel adaptive complementary ensemble empirical mode decomposition</article-title>. <source>Inf. Sci.</source> <volume>563</volume>, <fpage>269</fpage>&#x2013;<lpage>289</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2021.02.040</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumbalaparambi</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Menon</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Radhakrishnan</surname>
<given-names>V. P.</given-names>
</name>
<name>
<surname>Nair</surname>
<given-names>V. P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Assessment of urban air quality from Twitter communication using self-attention network and a multilayer classification model</article-title>. <source>Environ. Sci. Pollut. Res.</source> <volume>30</volume> (<issue>4</issue>), <fpage>10414</fpage>&#x2013;<lpage>10425</lpage>. <pub-id pub-id-type="doi">10.1007/s11356-022-22836-w</pub-id>
<pub-id pub-id-type="pmid">36074292</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lei</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Siu</surname>
<given-names>S. W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Application of ANN, XGBoost, and other ml methods to forecast air quality in Macau</article-title>. <source>Sustainability</source> <volume>15</volume> (<issue>6</issue>), <fpage>5341</fpage>. <pub-id pub-id-type="doi">10.3390/su15065341</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Air quality forecasting with artificial intelligence techniques: a scientometric and content analysis</article-title>. <source>Environ. Model. and Softw.</source> <volume>149</volume>, <fpage>105329</fpage>. <pub-id pub-id-type="doi">10.1016/j.envsoft.2022.105329</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ke</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Airformer: predicting nationwide air quality in China with transformers</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>37</volume> (<issue>12</issue>), <fpage>14329</fpage>&#x2013;<lpage>14337</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v37i12.26676</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Intelligent modeling strategies for forecasting air quality time series: a review</article-title>. <source>Appl. Soft Comput.</source> <volume>102</volume>, <fpage>106957</fpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2020.106957</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mishra</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Comparative analysis of Air Quality Index prediction using deep learning algorithms</article-title>. <source>Spatial Inf. Res.</source> <volume>32</volume> (<issue>1</issue>), <fpage>63</fpage>&#x2013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1007/s41324-023-00541-1</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharma</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Ghosh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mishra</surname>
<given-names>V. N.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Spatio-temporal Variations and Forecast of PM2. 5 concentration around selected Satellite Cities of Delhi, India using ARIMA model</article-title>. <source>Phys. Chem. Earth, Parts A/B/C</source> <volume>138</volume>, <fpage>103849</fpage>. <pub-id pub-id-type="doi">10.1016/j.pce.2024.103849</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A novel hybrid prediction model of air quality index based on variational modal decomposition and CEEMDAN-SE-GRU</article-title>. <source>Process Saf. Environ. Prot.</source> <volume>191</volume>, <fpage>2572</fpage>&#x2013;<lpage>2588</lpage>. <pub-id pub-id-type="doi">10.1016/j.psep.2024.10.018</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Attention is all you need</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>30</volume>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A spatiotemporal XGBoost model for PM2. 5 concentration prediction and its application in Shanghai</article-title>. <source>Heliyon</source> <volume>9</volume> (<issue>12</issue>), <fpage>e22569</fpage>. <pub-id pub-id-type="doi">10.1016/j.heliyon.2023.e22569</pub-id>
<pub-id pub-id-type="pmid">38058450</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An ensemble LSTM-based AQI forecasting model with decomposition-reconstruction technique via CEEMDAN and fuzzy entropy</article-title>. <source>Air Qual. Atmos. and Health</source> <volume>15</volume> (<issue>12</issue>), <fpage>2299</fpage>&#x2013;<lpage>2311</lpage>. <pub-id pub-id-type="doi">10.1007/s11869-022-01252-6</pub-id>
<pub-id pub-id-type="pmid">36196368</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Short-term prediction of PM2. 5 concentration by hybrid neural network based on sequence decomposition</article-title>. <source>Plos one</source> <volume>19</volume> (<issue>5</issue>), <fpage>e0299603</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0299603</pub-id>
<pub-id pub-id-type="pmid">38728371</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>A Graph Neural Network and Transformer-based model for PM2. 5 prediction through spatiotemporal correlation</article-title>. <source>Environ. Model. and Softw.</source> <volume>191</volume>, <fpage>106501</fpage>. <pub-id pub-id-type="doi">10.1016/j.envsoft.2025.106501</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zaini</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Ean</surname>
<given-names>L. W.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Malek</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A systematic literature review of deep learning neural network for time series air quality forecasting</article-title>. <source>Environ. Sci. Pollut. Res.</source> <volume>29</volume>, <fpage>4958</fpage>&#x2013;<lpage>4990</lpage>. <pub-id pub-id-type="doi">10.1007/s11356-021-17442-1</pub-id>
<pub-id pub-id-type="pmid">34807385</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Modeling air quality PM2. 5 forecasting using deep sparse attention-based transformer networks</article-title>. <source>Int. J. Environ. Sci. Technol.</source> <volume>20</volume> (<issue>12</issue>), <fpage>13535</fpage>&#x2013;<lpage>13550</lpage>. <pub-id pub-id-type="doi">10.1007/s13762-023-04900-1</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>PM2. 5 concentration prediction using weighted CEEMDAN and improved LSTM neural network</article-title>. <source>Environ. Sci. Pollut. Res.</source> <volume>30</volume> (<issue>30</issue>), <fpage>75104</fpage>&#x2013;<lpage>75115</lpage>. <pub-id pub-id-type="doi">10.1007/s11356-023-27630-w</pub-id>
<pub-id pub-id-type="pmid">37213020</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Graph neural networks: a review of methods and applications</article-title>. <source>AI open</source> <volume>1</volume>, <fpage>57</fpage>&#x2013;<lpage>81</lpage>. <pub-id pub-id-type="doi">10.1016/j.aiopen.2021.01.001</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>