<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Energy Res.</journal-id>
<journal-title>Frontiers in Energy Research</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Energy Res.</abbrev-journal-title>
<issn pub-type="epub">2296-598X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1346000</article-id>
<article-id pub-id-type="doi">10.3389/fenrg.2024.1346000</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Energy Research</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Short-term wind power forecasting based on dual attention mechanism and gated recurrent unit neural network</article-title>
<alt-title alt-title-type="left-running-head">Xu et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenrg.2024.1346000">10.3389/fenrg.2024.1346000</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Wu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Yang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2589671/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Fan</surname>
<given-names>Xinhao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Shen</surname>
<given-names>Zhifang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wu</surname>
<given-names>Qingchang</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Electrical and Information Technology</institution>, <institution>Yunnan Minzu University</institution>, <addr-line>Kunming</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Yunnan Key Laboratory of Unmanned Autonomous System</institution>, <addr-line>Kunming</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Lancang-Mekong International Vocational Institute</institution>, <addr-line>Kunming</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/865733/overview">Zijun Zhang</ext-link>, City University of Hong Kong, Hong Kong SAR, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2579708/overview">Zhongda Tian</ext-link>, Shenyang University of Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/615262/overview">Kenneth E. Okedu</ext-link>, Melbourne Institute of Technology, Australia</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1051445/overview">Yang Luoxiao</ext-link>, City University of Hong Kong, Hong Kong SAR, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Qingchang Wu, <email>wuqingchang@ymu.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>18</day>
<month>01</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1346000</elocation-id>
<history>
<date date-type="received">
<day>29</day>
<month>11</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>01</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Xu, Liu, Fan, Shen and Wu.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Xu, Liu, Fan, Shen and Wu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Accurate wind power forecasting is essential for both optimal grid scheduling and the massive absorption of wind power into the grid. However, the continuous changes in the contribution of various meteorological features to the forecasting of wind power output under different time or weather conditions, and the overlapping of wind power sequence cycles, make forecasting challenging. To address these problems, a short-term wind power forecasting model is established that integrates a gated recurrent unit (GRU) network with a dual attention mechanism (DAM). To compute the contributions of different features in real time, historical wind power data and meteorological information are first extracted using a feature attention mechanism (FAM). The feature sequences collected by the FAM are then used by the GRU network for preliminary forecasting. Subsequently, one-dimensional convolution employing several distinct convolution kernels is used to filter the GRU outputs. In addition, a multi-head time attention mechanism (MHTAM) is proposed and a Gaussian bias is introduced to assign different weights to different time steps of each modality. The final forecast results are produced by combining the outputs of the MHTAM. The results of the simulation experiment show that for 5-h, 10-h, and 20-h short-term wind power forecasting, the established&#xc2; DAM-GRU model performs better than comparative models on the basis of Root Mean Square Error (RMSE), Mean Absolute Error (MAE), R-squared (<italic>R</italic>
<sup>2</sup>), Square sum error (SSE), Mean absolute percentile error (MAPE), and Relative root mean square error (RRMSE) index.</p>
</abstract>
<kwd-group>
<kwd>wind power forecasting</kwd>
<kwd>feature attention mechanism</kwd>
<kwd>one-dimensional convolution</kwd>
<kwd>multi-head temporal attention mechanism</kwd>
<kwd>GRU</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Smart Grids</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The development and adoption of clean energy have a positive impact on protecting the environment, maintaining ecological balance, and reducing dependence on finite natural resources (<xref ref-type="bibr" rid="B8">Giebel and Kariniotakis 2017</xref>; <xref ref-type="bibr" rid="B32">Wang et al., 2021</xref>). Wind power, as a clean energy source, is steadily gaining prominence in the power grid. To meet the growing demand for electricity and achieve renewable energy goals, an increasing number of wind power projects worldwide are being connected to the electrical grid (<xref ref-type="bibr" rid="B31">Wang et al., 2022</xref>). However, the high unpredictability and volatility of wind energy can cause fluctuations in the frequency and voltage of the power system, which can be detrimental to the stability and quality of power. This simultaneously creates serious difficulties in scheduling and optimizing the grid (<xref ref-type="bibr" rid="B5">Duan et al., 2021</xref>).</p>
<p>The medium and long-term wind power forecasting is mainly to predict the annual and monthly power generation of wind farms to formulate power generation expectations and maintenance plans. Generally speaking, the results of short-term wind power forecasting have higher credibility than those of medium and long-term forecasting, so they can provide a good basis for grid scheduling, thus improving the ability of clean energy consumption. Therefore, improving the accuracy of wind power forecasting can provide an effective basis for grid scheduling, which is of great importance for the integration of large-scale wind power into the grid (<xref ref-type="bibr" rid="B2">Altan et al., 2021</xref>; <xref ref-type="bibr" rid="B4">Couto and Estanqueiro 2022</xref>).</p>
<p>There are two main approaches to wind power forecasting: statistical methods and methods based on physical principles. Physical methods use atmospheric science principles and meteorological data to estimate wind energy (<xref ref-type="bibr" rid="B24">Tian 2020</xref>). These physical methods are often employed for medium to long-term wind power forecasting, but short-term wind power forecasting heavily relies on conventional statistical approaches, such as exponential smoothing and time series analysis methods like the ARIMA model. These techniques generate short-term forecasts by analysing statistical features found in historical wind power data. However, wind power fluctuates a lot, and conventional models have a hard time explaining these intricate patterns (<xref ref-type="bibr" rid="B16">Liu et al., 2022</xref>). AI technologies, with deep learning as a prominent example, possess strong pattern recognition and data processing capabilities. Large-scale meteorological data and historical wind power generation data can be easily handled by them, improving the precision and dependability of power forecasting (<xref ref-type="bibr" rid="B36">Yang et al., 2021</xref>; <xref ref-type="bibr" rid="B20">Santhosh et al., 2018</xref>). Currently, ANN (<xref ref-type="bibr" rid="B38">Zhang et al., 2020</xref>), RNN <xref ref-type="bibr" rid="B10">Huang et al., 2021</xref>), and SVM (<xref ref-type="bibr" rid="B26">Tian and Chen, 2021a</xref>) are widely applied in time series forecasting tasks. Improved versions of RNN models are LSTM and GRU. They have achieved higher forecasting accuracy by addressing the issues of vanishing gradients and exploding gradients (<xref ref-type="bibr" rid="B27">Tian and Chen 2021b</xref>). LSTM is suitable for handling long-term dependencies but has a larger number of parameters, while GRU models are easier to train and achieve similar forecasting performance with fewer parameters (<xref ref-type="bibr" rid="B15">Liu et al., 2021</xref>; <xref ref-type="bibr" rid="B19">Saini et al., 2020</xref>). Therefore, in recent years, there has been an increasing amount of research on topics related to wind power forecasting using the GRU neural network as a basic model.</p>
<p>In <xref ref-type="bibr" rid="B12">Lin et al. (2021)</xref>, gray correlation analysis was employed to select similar days. Subsequently, the data was inputted into a GRU model for wind power forecasting. This approach, compared to models like Autoregressive Integrated Moving Average (ARIMA), enhances forecasting accuracy. However, the method ignores the contribution of meteorological features in the historical data to the wind power output for the time period to be forecasted. In <xref ref-type="bibr" rid="B33">Xiao et al. (2023)</xref>, the authors initially employ Weighted Principal Component Analysis (WPCA) with feature-weighted coefficients to reduce the dimensionality of wind power features. Subsequently, a GRU network optimized using the PSO algorithm is used for forecasting. The study considered the contribution of meteorological features to forecasts, but the contribution of individual meteorological features to forecasts varied over time and under different meteorological conditions. This shortcoming is remedied by <xref ref-type="bibr" rid="B11">Huang et al. (2023)</xref>, the authors consider the spatiotemporal correlation among adjacent wind turbines. They initially reconstructed wind power data from 24 surrounding wind turbines and organized it into a three-dimensional matrix. They then use a combination of three-dimensional CNN and GRU models for forecasting. Pre-processing meteorological information and using historical power data to train the GRU network can improve forecasting accuracy by optimising the model to better capture the relationship between power and meteorology (<xref ref-type="bibr" rid="B6">Farah et al., 2022</xref>; <xref ref-type="bibr" rid="B22">Sun et al., 2023</xref>). In addition to the processing and extracting of meteorological features, it is equally important to consider sequence autocorrelation from the time series perspective. There are also related scholars conducting research in this area.</p>
<p>Attentional mechanisms have breathed new life into the field of natural language processing and have been widely used in time series forecasting tasks, where they have been shown to help forecasting models extract key information (<xref ref-type="bibr" rid="B37">Zhang et al. 2021</xref>). In <xref ref-type="bibr" rid="B35">Yang and Zhang (2021)</xref>, the authors first use a Deep Attention Convolutional Recurrent Network (DACRN) to extract the features, then reconstruct the features using the developed auto-update memory module, then pattern cluster the feature reconstruction results using the K-shape clustering algorithm, and finally use the final prediction layer to predict the wind speed and experimentally validate the sophistication of the developed model. In <xref ref-type="bibr" rid="B3">Chi and Yang (2023)</xref>, the authors utilized Wavelet Transform (WT) to eliminate noise from the sample data. Subsequently, they employed a combination of Temporal Attention Mechanism and Bidirectional Gated Recurrent Units (BiGRU) to model the data. Finally, the model&#x2019;s performance was improved by utilizing a Time Convolutional Neural Network (TCN) to extract high-level temporal data. Trials have shown that this strategy improves forecasting accuracy. It is worth noting that the above two studies optimise and refine the baseline forecasting model in terms of meteorological features and time-series features respectively. However, the authors neglected the following three issues: first, the feature processing of both meteorological and time-series aspects are not well combined; second, the softmax function in the attention mechanism can achieve this effect well, but the original attention mechanism lacks the distinction of the relative position of the data in the time-series, which may require us to improve it when we use it for time-series feature extraction; Thirdly, it is possible that a single model could face some problems in extracting complex sequences directly, but data decomposition strategies were not considered for inclusion in the model.</p>
<p>Wind power data exhibits strong volatility, and achieving satisfactory accuracy through direct forecasting can be challenging. Therefore, decomposing wind power data and modeling forecasts separately for each component can effectively address this issue (<xref ref-type="bibr" rid="B23">Sun and Zhao 2020</xref>). In <xref ref-type="bibr" rid="B9">He and Wang (2021)</xref>, the authors employ EEMD to decompose wind power time series into easily analyzable subseries. The LASSO-QRNN model is then used for forecasting. Finally, researchers use the Kernel Density Estimation (KDE) method for post-processing to obtain more accurate short-term wind power forecasting. In <xref ref-type="bibr" rid="B1">Abdoos (2016)</xref>, the authors used Variable Modal Decomposition (VMD) to decompose the wind power series into different modes, and selected features using Gram-Schmidt Orthogonalisation (GSO), then used Extreme Learning Machine (ELM) to forecast the power of each mode, and finally superimposed the forecast results of each mode to obtain the final forecasting results. The data decomposition strategy can reduce the volatility of the time series and reduce the difficulty of forecasting for each model (<xref ref-type="bibr" rid="B28">Tian et al., 2020</xref>). However, in this strategy, the frequency of each modality is different, and it is difficult for the model to effectively integrate the meteorological features and better balance the importance of meteorological features and time series features.</p>
<p>Based on the above research results, this paper combines the advantages of the attention mechanism and the GRU network and improves them for the temporal attention mechanism, while using the idea of time series decomposition to better combine the two, and proposes a novel short-term wind power forecasting method which integrates the dual attention mechanism and GRU network. This method selects features in actual time by inputting power data and historical weather information into a feature attention layer. A GRU network is then used to make an initial forecasting. The output of the GRU network is then filtered using a number of one-dimensional convolutions with different kernel sizes. Ultimately, a MHTAM is used to assign weights to different time steps of each mode in a differential manner, and then the outputs are overlaid corresponding to the time steps to obtain the final forecast results.</p>
<p>In detail, the main contributions of the DAM-GRU model proposed in this paper to the study of short-term wind power forecasting are as follows:<list list-type="simple">
<list-item>
<p>1. Introduction of a feature attention mechanism that uses attention mechanisms to selectively extract historical meteorological or power features with higher contributions to the target forecasting time, while suppressing irrelevant features.</p>
</list-item>
<list-item>
<p>2. The model can better understand the characteristics and patterns of the wind power generation sequence by breaking it down and using one-dimensional convolution filtering with varying kernel sizes. This reduces the complexity of the sequence.</p>
</list-item>
<list-item>
<p>3. Introduction of a multi-head temporal attention mechanism to allocate varying weights to data at different time steps from multiple channels. This mechanism allows the model to simultaneously process information from different perspectives, rather than being confined to a single viewpoint, enabling the model to have a more comprehensive understanding of the temporal sequence structure.</p>
</list-item>
</list>
</p>
<p>The remaining portions of the paper are arranged as follows. The GRU model, the FAM, and the MHTAM are introduced in <xref ref-type="sec" rid="s2">Section 2</xref>. The suggested integrated structure, the specific procedure, and the assessment metrics are all provided in <xref ref-type="sec" rid="s3">Section 3</xref>. Comparing the merged model with other models and the model parameter settings, <xref ref-type="sec" rid="s4">Section 4</xref> examines the experimental outcomes. The study is finally summarized in <xref ref-type="sec" rid="s5">Section 5</xref>.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 Feature attention mechanism</title>
<p>Feature selection in wind power forecasting is a crucial step for improving model performance <xref ref-type="bibr" rid="B18">Meng et al. (2022)</xref>. Correlation analysis methods like Pearson correlation coefficient can assist in selecting meteorological features to enhance forecasting accuracy (<xref ref-type="bibr" rid="B13">Liu et al., 2017</xref>). However, the Pearson correlation coefficient method overlooks the changing contributions of meteorological features to forecasts over time. The presence of the softmax function gives the attention mechanism excellent time selection capabilities. We introduce a feature attention mechanism to extract crucial features at different time points in real-time, while suppressing irrelevant features. This improves the adaptability and accuracy of the model. <xref ref-type="fig" rid="F1">Figure 1</xref> depicts the basic idea behind the feature attention method used in this investigation.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The feature attention mechanism schematic diagram.</p>
</caption>
<graphic xlink:href="fenrg-12-1346000-g001.tif"/>
</fig>
<p>The feature attention mechanism first calculates the similarity between wind power and meteorological features at the same time point to obtain attention scores for each element. Taking time step t as an example, the calculation of its attention score is shown in Eq. <xref ref-type="disp-formula" rid="e1">1</xref>.<disp-formula id="e1">
<mml:math id="m1">
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m2">
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> represents the feature input vector at time step t, which includes five features for that moment: power, wind speed, wind direction, temperature, and air density; <italic>v</italic>
<sub>
<italic>e</italic>
</sub> and <italic>w</italic>
<sub>
<italic>e</italic>
</sub> is the weight. <italic>b</italic>
<sub>e</sub> is the bias term, and <italic>e</italic>
<sub>
<italic>t</italic>
</sub> encapsulates the attention score information for the five features at time step t.</p>
<p>Eq. <xref ref-type="disp-formula" rid="e2">2</xref> displays the probability distribution for the corresponding attention scores that are calculated using the softmax function.<disp-formula id="e2">
<mml:math id="m3">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mi>exp</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf2">
<mml:math id="m4">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> contains the importance degree corresponding to the m-th feature at the moment t. Finally, based on this importance degree and the input features, the feature attention output <italic>s</italic>
<sub>
<italic>t</italic>
</sub> is obtained, as shown in Eq. <xref ref-type="disp-formula" rid="e3">3</xref>.<disp-formula id="e3">
<mml:math id="m5">
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-2">
<title>2.2 Multi-head temporal attention mechanism</title>
<p>The influence of historical data with different values and positions on the forecasting point varies. Attention mechanisms can capture these relationships. However, standard self-attention mechanisms lack control over the positional relationships in time series data. This results in assigning similar importance to historical data with different relative positions (<xref ref-type="bibr" rid="B21">Shih et al., 2019</xref>). To address these issues, this paper adopts the method proposed in reference (<xref ref-type="bibr" rid="B34">Yang et al., 2018</xref>) with some modifications. One of the modifications includes incorporating a Gaussian bias into the attention mechanism. This modification ensures that varying weights are assigned to data for each time steps. The Gaussian bias assigns different weights to attention scores for different positions, following a Gaussian distribution. The center position of the Gaussian function is automatically adjusted during parameter learning to focus on the region that is highly influenced by historical information for the current forecasting value.</p>
<p>The temporal attention mechanism is illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>, with the input being the one-dimensional convolutional output sequence <inline-formula id="inf3">
<mml:math id="m6">
<mml:msup>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>.</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. The calculation of attention scores in the temporal attention layer is described in Eq. <xref ref-type="disp-formula" rid="e4">4</xref>.<disp-formula id="e4">
<mml:math id="m7">
<mml:mi>c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(4)</label>
</disp-formula>where <italic>V</italic>
<sub>
<italic>c</italic>
</sub> and <italic>W</italic>
<sub>
<italic>c</italic>
</sub> are weight matrices; <italic>b</italic>
<sub>
<italic>c</italic>
</sub> represents the bias term, and the attention score <inline-formula id="inf4">
<mml:math id="m8">
<mml:mi>c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The temporal attention mechanism schematic diagram.</p>
</caption>
<graphic xlink:href="fenrg-12-1346000-g002.tif"/>
</fig>
<p>To achieve differential weight allocation across different time steps, Gaussian bias and attention scores are jointly input into the softmax function to compute attention probabilities <italic>&#x3b1;</italic>
<sub>
<italic>t</italic>
</sub>, as shown in Eq. <xref ref-type="disp-formula" rid="e5">5</xref>.<disp-formula id="e5">
<mml:math id="m9">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>max</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(5)</label>
</disp-formula>where <italic>w</italic>
<sub>
<italic>&#x3b1;</italic>
</sub> represents a weight factor; <italic>&#x3b1;</italic>
<sub>
<italic>t</italic>
</sub> represents the probability of the attention score at time step t, and <italic>G</italic>
<sub>
<italic>t</italic>
</sub> represents the Gaussian bias at time step t. <italic>G</italic>
<sub>
<italic>t</italic>
</sub> reflects the degree of closeness between the current moment and the central position moment, and its calculation method is as follows:<disp-formula id="e6">
<mml:math id="m10">
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m11">
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="array">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mi>D</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="array">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mi>z</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m12">
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="e9">
<mml:math id="m13">
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(9)</label>
</disp-formula>where <italic>Q</italic>
<sub>
<italic>t</italic>
</sub> represents the center position at time step t, and its value is ultimately determined based on the parameter <italic>u</italic>
<sub>
<italic>t</italic>
</sub>, which is learned based on the value of t; <italic>&#x3c3;</italic> is set to <inline-formula id="inf5">
<mml:math id="m14">
<mml:mfrac>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:math>
</inline-formula>, where D is the window size for this mode, each mode has a separate window to define its window range, with a larger value indicating a longer sequence related to the current time step; I is a real number that ranges from 0 to the input sequence&#x2019;s length; <italic>v</italic>
<sub>
<italic>q</italic>
</sub>, <italic>v</italic>
<sub>
<italic>z</italic>
</sub>, <italic>w</italic>
<sub>
<italic>q</italic>
</sub> and <italic>w</italic>
<sub>
<italic>z</italic>
</sub> are weight coefficients, and z is the scalar factor for selecting the window for this mode; <inline-formula id="inf6">
<mml:math id="m15">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> in the original reference represents the key-value relationship between semantics. Here, it is set to be the same as the kernel size k of the one-dimensional convolution. The underlying idea is that if a mode has a larger convolution kernel, its window will be larger. This allows different sequences to have different linear characteristics, aiding the model in capturing trend components of time series data.</p>
<p>Eq. <xref ref-type="disp-formula" rid="e10">10</xref> illustrates how the output of the temporal attention layer is finally calculated based on the probability <italic>&#x3b1;</italic>
<sub>
<italic>i</italic>
</sub> of the temporal attention scores.<disp-formula id="e10">
<mml:math id="m16">
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munder>
</mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>A single temporal attention mechanism is responsible for capturing the temporal weights of a single channel, while the multi-head temporal attention mechanism simply combines them. Applying one-dimensional convolutions with multiple diverse kernels to filter the GRU network&#x2019;s output is necessary to guarantee that each temporal attention mechanism can extract distinct patterns of temporal information. When the length of the convolution kernel is k, the convolution formula is as shown in Eq. <xref ref-type="disp-formula" rid="e11">11</xref>.<disp-formula id="e11">
<mml:math id="m17">
<mml:msup>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>u</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf7">
<mml:math id="m18">
<mml:msubsup>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> represents the <italic>i</italic>th element of the feature map; <inline-formula id="inf8">
<mml:math id="m19">
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>u</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> represents the activation function. <italic>M</italic>
<sub>
<italic>H</italic>
</sub> represents the spatial extent of the convolution kernel; <italic>w</italic>
<sub>
<italic>l</italic>
</sub> is the corresponding weight; <italic>h</italic>
<sub>
<italic>il</italic>
</sub> represents the <italic>l</italic>th element in the input data with i as the center; <italic>b</italic>
<sub>
<italic>l</italic>
</sub> is the bias of the convolution kernel. After the convolution output undergoes feature extraction through the temporal attention mechanism, the results are summed up with step-wise weighting, as shown in Eq. <xref ref-type="disp-formula" rid="e12">12</xref>.<disp-formula id="e12">
<mml:math id="m20">
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:math>
<label>(12)</label>
</disp-formula>where the projected output of the model at time step t is denoted by <italic>p</italic>
<sub>
<italic>t</italic>
</sub>; The output of the <italic>i</italic>th temporal attention head at time step t is denoted by <inline-formula id="inf9">
<mml:math id="m21">
<mml:msubsup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>; <italic>w</italic>
<sub>
<italic>p</italic>
</sub> corresponds to the weight associated with it. H represents the total number of attention heads.</p>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> shows the details of the implementation of the multi-head temporal attention mechanism.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Architecture of multi-head temporal attention mechanism.</p>
</caption>
<graphic xlink:href="fenrg-12-1346000-g003.tif"/>
</fig>
</sec>
<sec id="s2-3">
<title>2.3 GRU</title>
<p>Compared to the complex LSTM, the GRU has a simpler structure and higher computational efficiency. In this paper, the GRU network is chosen as the main component to construct the model. The feature selection and forgetting functions of GRU are implemented by only reset gates and update gates. GRU is relatively more efficient in terms of computational efficiency and number of parameters due to its simple structure. The update gate determines how much of the past information is retained, and the parameter values of the update gate are learned through training thus allowing the GRU unit to dynamically capture long-term dependencies in the sequence. The reset gate controls the inflow of historical information into the candidate hidden state and thus determines whether to disregard past information (<xref ref-type="bibr" rid="B30">van Heerden et al., 2022</xref>). Its internal structure is depicted in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Internal structure of GRU unit.</p>
</caption>
<graphic xlink:href="fenrg-12-1346000-g004.tif"/>
</fig>
<p>Taking moment t as an example, the inputs to the gated loop unit are <italic>s</italic>
<sub>
<italic>t</italic>
</sub> and the hidden state <italic>h</italic>
<sub>
<italic>t</italic>&#x2212;1</sub> from the previous moment. Inputs are processed to calculate the outputs of the update gate and reset gate, which are represented by Eqs <xref ref-type="disp-formula" rid="e13">13</xref>, <xref ref-type="disp-formula" rid="e14">14</xref>.<disp-formula id="e13">
<mml:math id="m22">
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(13)</label>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m23">
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(14)</label>
</disp-formula>where <italic>w</italic>
<sub>
<italic>sr</italic>
</sub>, <italic>w</italic>
<sub>
<italic>hr</italic>
</sub>, <italic>w</italic>
<sub>
<italic>sz</italic>
</sub>and <italic>w</italic>
<sub>
<italic>hz</italic>
</sub> represent weight terms; <italic>r</italic>
<sub>
<italic>t</italic>
</sub> is the output of the reset gate, and <italic>z</italic>
<sub>
<italic>t</italic>
</sub> is the output of the update gate; <italic>&#x3c3;</italic> represents the sigmoid function.</p>
<p>The reset gate is used to determine the significance of the output from the previous time step. It combines this information with the current input to calculate the current hidden state, as shown in Eq. <xref ref-type="disp-formula" rid="e15">15</xref>.<disp-formula id="e15">
<mml:math id="m24">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2297;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(15)</label>
</disp-formula>where <italic>w</italic>
<sub>
<italic>sh</italic>
</sub> and <italic>w</italic>
<sub>
<italic>hh</italic>
</sub> represent the weight terms for the importance at time step <italic>t</italic> and time step <italic>t</italic> &#x2212; 1, respectively. <inline-formula id="inf10">
<mml:math id="m25">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> signifies the hidden state at time step <italic>t</italic>, and &#x2297; denotes the Hadamard product.</p>
<p>Eq. <xref ref-type="disp-formula" rid="e16">16</xref> is the final calculation used to determine the output of the GRU at the current time.<disp-formula id="e16">
<mml:math id="m26">
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2297;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2297;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(16)</label>
</disp-formula>
</p>
<p>According to Eq. <xref ref-type="disp-formula" rid="e16">16</xref>, when the update gate output <italic>z</italic>
<sub>
<italic>t</italic>
</sub> approaches 0, the output of the GRU unit is mainly determined by the output of the previous time step. Conversely, when <italic>z</italic>
<sub>
<italic>t</italic>
</sub> approaches 1, it is primarily determined by the hidden state of the current time step (<xref ref-type="bibr" rid="B14">Liu et al., 2023</xref>).</p>
</sec>
</sec>
<sec id="s3">
<title>3 Model design</title>
<sec id="s3-1">
<title>3.1 DAM-GRU model calculation process</title>
<p>Wind power is subject to weather conditions that are highly random and volatile. In order to decompose and predict wind power generation while considering climate features comprehensively and avoiding situations where historical information is insufficiently learned, such as forecasting lag, we establish a forecasting model that combines feature-based temporal dual attention mechanisms with GRU networks. The complete structure of the model is depicted in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>DAM-GRU model architecture.</p>
</caption>
<graphic xlink:href="fenrg-12-1346000-g005.tif"/>
</fig>
<p>The DAM-GRU model-based short-term wind power forecasting procedure is as follows:</p>
<p>1) Normalize the power, wind speed, temperature, and air density features using Eq. <xref ref-type="disp-formula" rid="e17">17</xref>.<disp-formula id="e17">
<mml:math id="m27">
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">norm</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(17)</label>
</disp-formula>where <italic>u</italic>
<sub>norm</sub> represents the normalized result. The text represents the values of the four mentioned feature sequences. The maximum and minimum values in the raw data are denoted by <italic>u</italic>
<sub>max</sub> and <italic>u</italic>
<sub>min</sub>, respectively. Normalize the wind direction feature using Eq. <xref ref-type="disp-formula" rid="e18">18</xref>.<disp-formula id="e18">
<mml:math id="m28">
<mml:msubsup>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">norm</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(18)</label>
</disp-formula>where <inline-formula id="inf11">
<mml:math id="m29">
<mml:msubsup>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">norm</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> represents the normalized result of the wind direction feature, and <italic>u</italic>
<sup>(3)</sup> represents the wind direction feature sequence.</p>
<p>2) The FAM assigns different weights to meteorological features based on different weather conditions. This allows it to dynamically extract key features in real time, improving the forecasting process.</p>
<p>3) Making initial predictions using a two-layer GRU model based on the extracted features.</p>
<p>4) One-dimensional convolutional models with H convolutional kernels of different sizes are used to apply convolutional filtering operations to the initial forecasting results of the GRU model. This is done to extract the timing patterns of wind power for different time series and reduce the complexity of individual timings.</p>
<p>5) In order to avoid forecasting lag, assign temporal feature weights to the H distinct periodic patterns independently using a multi-head temporal attention technique.</p>
<p>6) Combine the predictions of each sub-sequence by performing a weighted summation. Reverse the normalization process to obtain the final forecast result.</p>
</sec>
<sec id="s3-2">
<title>3.2 Evaluation metrics</title>
<p>In order to statistically assess the accuracy of the wind power forecasting model, we compare RMSE, MAE, <italic>R</italic>
<sup>2</sup> SSE, MAPE, and RRMSE using the following formulas:<disp-formula id="e19">
<mml:math id="m30">
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:math>
<label>(19)</label>
</disp-formula>
<disp-formula id="e20">
<mml:math id="m31">
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(20)</label>
</disp-formula>
<disp-formula id="e21">
<mml:math id="m32">
<mml:msup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(21)</label>
</disp-formula>
<disp-formula id="e22">
<mml:math id="m33">
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
<label>(22)</label>
</disp-formula>
<disp-formula id="e23">
<mml:math id="m34">
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mi>%</mml:mi>
</mml:math>
<label>(23)</label>
</disp-formula>
<disp-formula id="e24">
<mml:math id="m35">
<mml:mi>R</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:math>
<label>(24)</label>
</disp-formula>where <inline-formula id="inf12">
<mml:math id="m36">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> denotes the average value of the actual value sequence, n represents the number of forecast steps, <italic>p</italic>
<sub>
<italic>i</italic>
</sub> and <inline-formula id="inf13">
<mml:math id="m37">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> represent the predicted and actual values for the <italic>i</italic>th sample point, respectively. Smaller RMSE, MAE SSE, MAPE, and RRMSE indicate better model performance, while an <italic>R</italic>
<sup>2</sup> closer to 1 indicates higher accuracy (<xref ref-type="bibr" rid="B25">Tian 2021</xref>; <xref ref-type="bibr" rid="B28">Tian et al. 2021</xref>).</p>
</sec>
</sec>
<sec id="s4">
<title>4 Case study</title>
<p>The experiment was conducted on a hardware platform consisting of a GeForce GTX 1050Ti GPU with 2 &#xd7; 8 GB DDR4 memory. The programming language used was Python 3.7, and the model was constructed using the TensorFlow 2.1 deep learning framework.</p>
<p>In order to validate the performance of the proposed model, power and meteorological data collected from a wind farm located in Inner Mongolia, China, were used for the experiments. The dataset covers the time range from January 1 to 30 June 2020, and the data is sampled every 15 min, resulting in a total of 17,568 data points. Where 14,000 data points are allocated for the training set, 250 data points are allocated for validation, and 3,318 data points are allocated for testing. The installed capacity of the wind farm is 100 MW. Each epoch consists of 80 batches, with a training batch size of 100 and a learning rate of 0.001. The training phase utilizes the Adam optimizer.</p>
<sec id="s4-1">
<title>4.1 Model parameter configuration</title>
<p>The forecasting model employs feature sequences of length 30 steps as input, meaning that wind power for future n time steps is predicted using the previous 30 steps of features. The number of FAM heads in the model is equal to the number of input time steps. A double-layer GRU network is employed for the preliminary forecasting of wind power, with each layer containing 64 GRU units. The choice of attention heads is crucial, and in this case, parameter tuning experiments are conducted under 20 steps forecasting horizon. The number of attention heads, denoted as H, is tested with values of 2, 3, 4, 5, and 6, and the fluctuation of errors under different attention head numbers is shown in <xref ref-type="table" rid="T1">Table 1</xref>. The minimum error is achieved when H is set to 4. The convolutional kernels corresponding to the 4 attention heads have sizes of 3 &#xd7; 1, 5 &#xd7; 1, 7 &#xd7; 1, and 9 &#xd7; 1, with a stride of 1, the edge padding strategy is set to &#x201c;same&#x201d;, and the activation function is set to &#x201c;ReLU&#x201d;. The number of time steps in which the model outputs its forecasts is the number of output time steps for each TAM.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Influence of the number of temporal attention heads on error.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Number of temporal attention heads (H)</th>
<th align="center">2</th>
<th align="center">3</th>
<th align="center">4</th>
<th align="center">5</th>
<th align="center">6</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">RMSE</td>
<td align="center">1.2124</td>
<td align="center">0.9891</td>
<td align="center">0.6809</td>
<td align="center">0.9023</td>
<td align="center">0.9201</td>
</tr>
<tr>
<td align="center">MAE</td>
<td align="center">0.8789</td>
<td align="center">0.7212</td>
<td align="center">0.4866</td>
<td align="center">0.6187</td>
<td align="center">0.6512</td>
</tr>
<tr>
<td align="center">
<italic>R</italic>
<sup>2</sup>
</td>
<td align="center">0.9893</td>
<td align="center">0.9919</td>
<td align="center">0.9939</td>
<td align="center">0.9935</td>
<td align="center">0.9927</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2">
<title>4.2 Validation of model effectiveness</title>
<p>In this paper, three core components of the DAM-GRU model are the FAM, the GRU network, and the MHTAM. In order to emphasize the diversity of each time feature, a Gaussian bias is added to the multi-head time attention module. To evaluate the influence of each component of the DAM-GRU model on forecasting performance, this section compares the model with other models that use the same dataset and GRU network parameters. The DAM-GRU models compared include the GRU model, GRU-TAM model, FAM-GRU model, and GRU-MHTAM model. The GRU-TAM model combines a temporal attention mechanism with a single head within the GRU network. To ensure a fair comparison in the experiments, a 3 &#xd7; 1 convolutional kernel is used to filter the initial forecasts made by the GRU. The FAM-GRU model combines the FAM with the GRU network, while the GRU-MHTAM model combines the GRU network with the multi-head temporal attention mechanism. The forecasting horizons are categorized into three levels: 20 steps (5-h), 40 steps (10-h), and 80 steps (20-h). Correspondingly, the lengths of the historical sequences are 30 steps, 60 steps, and 120 steps. This means that the model uses historical input sequences of 30, 60, and 120 steps to forecast power values for 20, 40, and 80 steps into the future, respectively.</p>
<p>
<xref ref-type="fig" rid="F6">Figure 6</xref> show the comparison of forecast curves and box plots for five models at forecasting durations of 5-h, 10-h, and 20-h. All three sets of forecast curves show that the single GRU forecast model has a significant lag. This lag is attributed to its limited ability to efficiently extract highly correlated meteorological and time series features. The inclusion of the FAM and single-head temporal attention mechanism results in varying degrees of improvement, as evident in three box plots. The incorporation of the MHTAM leads to a significant enhancement in model performance. The proposed model in this paper demonstrates excellent feature extraction capabilities, making it adaptable to the rapid fluctuations in wind power. Overall, it demonstrates a significant decrease in lag and improved alignment with the observed values compared to the other models. Overall, the errors in model forecasts increase rapidly with increasing forecast length, and certain models have varying degrees; of outlier forecasts. This implies that the models require additional features to account for variations in the meteorological data as the forecast time horizon increases. The forecasting process becomes more complex and challenging due to various external factors that can impact the output of wind power.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>5-h, 10-h and 20-h forecast curves and box plots for 5 models including GRU, GRU-TAM, FAM-GRU, GRU-MHTAM, and DAM-GRU.</p>
</caption>
<graphic xlink:href="fenrg-12-1346000-g006.tif"/>
</fig>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> shows a comparison of the predictive performance for different attention mechanisms at 5-h, 10-h, and 20-h. From the table, it can be observed that the forecasting errors of all models increase rapidly as the forecasting steps increase, while the goodness-of-fit indicator, the coefficient of determination (<italic>R</italic>
<sup>2</sup>), decreases. This suggests that longer-term sequences exhibit weaker regularity and higher complexity, making it more challenging for models to extract features. The performance of the forecasting model can be improved by incorporating temporal and feature attention mechanisms. The model can more easily adjust to the intricacy of wind power sequences because of these attention mechanisms, which provide it the flexibility to focus on various attributes and time steps. Taking the 20-h forecast results as an example, the GRU-TAM model reduced the RMSE, MAE, SSE, MAPE, and RRMSE error metrics by 26.3%, 18.5%, 45.6%, 19.1%, and 24.6% respectively, and increased the <italic>R</italic>
<sup>2</sup> by 3.4% compared to the GRU model. On the other hand, the FAM-GRU model reduced the RMSE, MAE, SSE, MAPE, and RRMSE errors metrics by 35.8%, 32.4%, 58.8%, 25.2%, and 33.8% respectively, and increased the <italic>R</italic>
<sup>2</sup> by 4.3% compared to the GRU model. This demonstrates the necessity of adding attention mechanisms to help the GRU model extract features and temporal information. The GRU-MHTAM model, compared to the GRU-TAM model, reduces the RMSE, MAE, SSE, MAPE, and RRMSE error metrics by 53.0%, 56.4%, 78.9%, 53.7%, and 52.8% respectively, and increases <italic>R</italic>
<sup>2</sup> by 3.0%. This suggests that the MHTAM can be effective in dealing with time series features. In this experiment, the parts of the DAM-GRU model are split and tested for comparison, which proves the validity of the model design method from two perspectives: feature and time series, and demonstrates that the DAM-GRU model has good forecasting performance.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Comparison of predictive performance with different attention mechanisms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Forecasting horizon</th>
<th rowspan="2" align="center">Model</th>
<th colspan="6" align="center">Evaluation metrics</th>
</tr>
<tr>
<th align="center">RMSE</th>
<th align="center">MAE</th>
<th align="center">
<italic>R</italic>
<sup>2</sup>
</th>
<th align="center">SSE</th>
<th align="center">MAPE</th>
<th align="center">RRMSE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="5" align="center">20 Steps (5 h)</td>
<td align="center">GRU</td>
<td align="center">2.2720</td>
<td align="center">1.8029</td>
<td align="center">0.9313</td>
<td align="center">103.2357</td>
<td align="center">25.2024</td>
<td align="center">0.3378</td>
</tr>
<tr>
<td align="center">GRU-TAM</td>
<td align="center">1.6755</td>
<td align="center">1.4691</td>
<td align="center">0.9626</td>
<td align="center">56.1443</td>
<td align="center">20.3853</td>
<td align="center">0.2546</td>
</tr>
<tr>
<td align="center">FAM-GRU</td>
<td align="center">1.4586</td>
<td align="center">1.2191</td>
<td align="center">0.9717</td>
<td align="center">42.5488</td>
<td align="center">16.8418</td>
<td align="center">0.2235</td>
</tr>
<tr>
<td align="center">GRU-MHTAM</td>
<td align="center">0.7880</td>
<td align="center">0.6408</td>
<td align="center">0.9919</td>
<td align="center">12.4186</td>
<td align="center">9.4291</td>
<td align="center">0.1201</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">0.6809</td>
<td align="center">0.4866</td>
<td align="center">0.9939</td>
<td align="center">9.2734</td>
<td align="center">5.5039</td>
<td align="center">0.0774</td>
</tr>
<tr>
<td rowspan="5" align="center">40 Steps (10 h)</td>
<td align="center">GRU</td>
<td align="center">3.2978</td>
<td align="center">2.5479</td>
<td align="center">0.8973</td>
<td align="center">435.0079</td>
<td align="center">16.3893</td>
<td align="center">0.2072</td>
</tr>
<tr>
<td align="center">GRU-TAM</td>
<td align="center">2.4142</td>
<td align="center">1.9265</td>
<td align="center">0.9450</td>
<td align="center">233.1439</td>
<td align="center">12.3113</td>
<td align="center">0.1529</td>
</tr>
<tr>
<td align="center">FAM-GRU</td>
<td align="center">2.0030</td>
<td align="center">1.6310</td>
<td align="center">0.9621</td>
<td align="center">160.4805</td>
<td align="center">11.2943</td>
<td align="center">0.1503</td>
</tr>
<tr>
<td align="center">GRU-MHTAM</td>
<td align="center">1.1418</td>
<td align="center">0.9254</td>
<td align="center">0.9877</td>
<td align="center">52.1516</td>
<td align="center">7.4974</td>
<td align="center">0.1030</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">0.8822</td>
<td align="center">0.6644</td>
<td align="center">0.9926</td>
<td align="center">31.1283</td>
<td align="center">6.7024</td>
<td align="center">0.0856</td>
</tr>
<tr>
<td rowspan="5" align="center">80 Steps (20 h)</td>
<td align="center">GRU</td>
<td align="center">4.1980</td>
<td align="center">3.2628</td>
<td align="center">0.8140</td>
<td align="center">1409.8770</td>
<td align="center">21.7050</td>
<td align="center">0.3153</td>
</tr>
<tr>
<td align="center">GRU-TAM</td>
<td align="center">3.1024</td>
<td align="center">2.4259</td>
<td align="center">0.8984</td>
<td align="center">769.9804</td>
<td align="center">14.0641</td>
<td align="center">0.1870</td>
</tr>
<tr>
<td align="center">FAM-GRU</td>
<td align="center">2.7326</td>
<td align="center">2.1432</td>
<td align="center">0.9212</td>
<td align="center">597.3869</td>
<td align="center">12.1121</td>
<td align="center">0.1557</td>
</tr>
<tr>
<td align="center">GRU-MHTAM</td>
<td align="center">1.8812</td>
<td align="center">1.4849</td>
<td align="center">0.9626</td>
<td align="center">283.1256</td>
<td align="center">8.8778</td>
<td align="center">0.1149</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">1.2521</td>
<td align="center">1.0003</td>
<td align="center">0.9835</td>
<td align="center">125.4133</td>
<td align="center">7.5521</td>
<td align="center">0.1062</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="table" rid="T3">Table 3</xref> shows a comparison of the time required for training and forecasting for the five models. The time required for training and prediction is not strictly increasing or decreasing due to complex factors such as GPU performance and initial parameters within the model, but generally shows some degree of regularity. From the table, it can be seen that the sum of training time for FAM-GRU and GRU-MHTAM is similar to the sum of training time for GRU and DAM-GRU, and the forecast time has the same pattern, which is consistent with the number of parameters of the model. As the number of forecast steps increases, the time required for training and forecasting may also increase, due to the fact that an increase in the number of forecast steps corresponds to an increase in the number of units in the input layer of the model.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Comparison of training time and forecasting time with different attention mechanisms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Forecasting horizon</th>
<th align="center">Model</th>
<th align="center">Training time (s)</th>
<th align="center">Forecast time (ms)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="5" align="center">20 Steps (5 h)</td>
<td align="center">GRU</td>
<td align="center">120</td>
<td align="center">16</td>
</tr>
<tr>
<td align="center">GRU-TAM</td>
<td align="center">135</td>
<td align="center">31</td>
</tr>
<tr>
<td align="center">FAM-GRU</td>
<td align="center">187</td>
<td align="center">84</td>
</tr>
<tr>
<td align="center">GRU-MHTAM</td>
<td align="center">248</td>
<td align="center">199</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">304</td>
<td align="center">262</td>
</tr>
<tr>
<td rowspan="5" align="center">40 Steps (10 h)</td>
<td align="center">GRU</td>
<td align="center">122</td>
<td align="center">17</td>
</tr>
<tr>
<td align="center">GRU-TAM</td>
<td align="center">146</td>
<td align="center">34</td>
</tr>
<tr>
<td align="center">FAM-GRU</td>
<td align="center">199</td>
<td align="center">83</td>
</tr>
<tr>
<td align="center">GRU-MHTAM</td>
<td align="center">253</td>
<td align="center">223</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">329</td>
<td align="center">264</td>
</tr>
<tr>
<td rowspan="5" align="center">80 Steps (20 h)</td>
<td align="center">GRU</td>
<td align="center">127</td>
<td align="center">22</td>
</tr>
<tr>
<td align="center">GRU-TAM</td>
<td align="center">171</td>
<td align="center">42</td>
</tr>
<tr>
<td align="center">FAM-GRU</td>
<td align="center">206</td>
<td align="center">97</td>
</tr>
<tr>
<td align="center">GRU-MHTAM</td>
<td align="center">289</td>
<td align="center">251</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">368</td>
<td align="center">277</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-3">
<title>4.3 Forecasting performance tests</title>
<p>In order to test whether the proposed DAM-GRU model has advanced prediction performance, we select CNN-GRU (<xref ref-type="bibr" rid="B7">Gao et al., 2023</xref>), VMD-CNN-GRU (<xref ref-type="bibr" rid="B39">Zhao et al. 2023</xref>), and MTTFA-GRU (<xref ref-type="bibr" rid="B17">Liu and Zhou, 2024</xref>) under the same data set algorithms as comparison models for the experiment. Among them, the number of input features of CNN-GRU and MTTFA-LSTM models is 5, the VMD-CNN-GRU model uses VMD decomposition to decompose the wind speed into four submodules, and then combines them with the wind power series using the CNN-GRU model to predict. To ensure the fairness of the experiments, the training batch size and iteration number of the comparison models are the same as the models in this paper, using the same settings.</p>
<p>At forecast horizons of 5 hours, 10 hours, and 20 hours, <xref ref-type="fig" rid="F7">Figure 7</xref> compare the suggested DAM-GRU model with three models. Box plots and forecast curves are compared in the figures. We find that the DAM-GRU model forecasts values that are quite similar to the observations. It demonstrates excellent predictive performance even during rapid changes in wind power over a short time frame. When combined with the accuracy metrics presented in <xref ref-type="table" rid="T4">Table 4</xref>, the DAM-GRU model outperforms the comparative models for forecasting horizons of 5 h, 10 h, and 20 h. Specifically, for a forecasting horizon of 5 h, the proposed model reduces RMSE, MAE, SSE, MAPE, and RRMSE forecasting errors by 6.0%, 18.9%, 11.7%, 20.6%, and 11.3% respectively, and increased the <italic>R</italic>
<sup>2</sup> by 0.8% compared to the MTTFA-LSTM model. For 10 and 20 h, The DAM-GRU model also outperforms the MTTFA-LSTM model to varying degrees. The DAM-GRU model&#x2019;s performance is also superior to that of the CNN-GRU and VMD-CNN-GRU models. The validity of the DAM-GRU model was further confirmed.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>5-h, 10-h and 20-h forecast curves and box plots for 4 models including CNN-GRU, VMD-CNN-GRU, MTTFA-LSTM, and DAM-GRU.</p>
</caption>
<graphic xlink:href="fenrg-12-1346000-g007.tif"/>
</fig>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Comparison of multi-stage forecasting errors with other classic GRU-based models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Forecasting horizon</th>
<th rowspan="2" align="center">Model</th>
<th colspan="6" align="center">Evaluation metrics</th>
</tr>
<tr>
<th align="center">RMSE</th>
<th align="center">MAE</th>
<th align="center">
<italic>R</italic>
<sup>2</sup>
</th>
<th align="center">SSE</th>
<th align="center">MAPE/%</th>
<th align="center">RRMSE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="center">20 Steps (5 h)</td>
<td align="center">CNN-GRU</td>
<td align="center">0.7893</td>
<td align="center">0.6405</td>
<td align="center">0.9919</td>
<td align="center">12.4586</td>
<td align="center">7.6498</td>
<td align="center">0.0982</td>
</tr>
<tr>
<td align="center">VMD-CNN-GRU</td>
<td align="center">0.7584</td>
<td align="center">0.6008</td>
<td align="center">0.9925</td>
<td align="center">11.5044</td>
<td align="center">7.6123</td>
<td align="center">0.0946</td>
</tr>
<tr>
<td align="center">MTTFA-LSTM</td>
<td align="center">0.7246</td>
<td align="center">0.5998</td>
<td align="center">0.9931</td>
<td align="center">10.5018</td>
<td align="center">6.9354</td>
<td align="center">0.0873</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">0.6809</td>
<td align="center">0.4866</td>
<td align="center">0.9939</td>
<td align="center">9.2734</td>
<td align="center">5.5039</td>
<td align="center">0.0774</td>
</tr>
<tr>
<td rowspan="4" align="center">40 Steps (10 h)</td>
<td align="center">CNN-GRU</td>
<td align="center">1.2375</td>
<td align="center">0.9935</td>
<td align="center">0.9855</td>
<td align="center">61.2598</td>
<td align="center">10.3721</td>
<td align="center">0.1887</td>
</tr>
<tr>
<td align="center">VMD-CNN-GRU</td>
<td align="center">1.1479</td>
<td align="center">0.9492</td>
<td align="center">0.9876</td>
<td align="center">52.7060</td>
<td align="center">8.9672</td>
<td align="center">0.1428</td>
</tr>
<tr>
<td align="center">MTTFA-LSTM</td>
<td align="center">1.0216</td>
<td align="center">0.8394</td>
<td align="center">0.9901</td>
<td align="center">41.7429</td>
<td align="center">7.0038</td>
<td align="center">0.0923</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">0.8822</td>
<td align="center">0.6644</td>
<td align="center">0.9926</td>
<td align="center">31.1283</td>
<td align="center">6.7024</td>
<td align="center">0.0856</td>
</tr>
<tr>
<td rowspan="4" align="center">80 Steps (20 h)</td>
<td align="center">CNN-GRU</td>
<td align="center">1.7151</td>
<td align="center">1.4209</td>
<td align="center">0.9690</td>
<td align="center">235.3202</td>
<td align="center">12.2857</td>
<td align="center">0.2571</td>
</tr>
<tr>
<td align="center">VMD-CNN-GRU</td>
<td align="center">1.5679</td>
<td align="center">1.3076</td>
<td align="center">0.9740</td>
<td align="center">196.6564</td>
<td align="center">1.1592</td>
<td align="center">0.2058</td>
</tr>
<tr>
<td align="center">MTTFA-LSTM</td>
<td align="center">1.4042</td>
<td align="center">1.1069</td>
<td align="center">0.9792</td>
<td align="center">157.7458</td>
<td align="center">8.8852</td>
<td align="center">0.1444</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">1.2521</td>
<td align="center">1.0003</td>
<td align="center">0.9835</td>
<td align="center">125.4133</td>
<td align="center">7.5511</td>
<td align="center">0.1062</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To compare the training and forecasting time of the proposed DAM-GRU model with other mainstream and advanced GRU-based models. <xref ref-type="table" rid="T5">Table 5</xref> shows that the proposed DAM-GRU model has a slight increase in training time and forecasting time compared to CNN-GRU, VMD-CNN-GRU, and MTTFA-LSTM models, but considering its obvious improvement in forecasting performance and that the time spent in actual online training is much less than the length of forecasting for each training, it can meet the technical requirements in practical applications.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Comparison of multi-stage training time and forecasting time with other classic GRU-based models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Forecasting horizon</th>
<th align="center">Model</th>
<th align="center">Training time (s)</th>
<th align="center">Forecast time (ms)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="center">20 Steps (5 h)</td>
<td align="center">CNN-GRU</td>
<td align="center">188</td>
<td align="center">116</td>
</tr>
<tr>
<td align="center">VMD-CNN-GRU</td>
<td align="center">253</td>
<td align="center">182</td>
</tr>
<tr>
<td align="center">MTTFA-LSTM</td>
<td align="center">286</td>
<td align="center">219</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">304</td>
<td align="center">262</td>
</tr>
<tr>
<td rowspan="4" align="center">40 Steps (10 h)</td>
<td align="center">CNN-GRU</td>
<td align="center">206</td>
<td align="center">117</td>
</tr>
<tr>
<td align="center">VMD-CNN-GRU</td>
<td align="center">281</td>
<td align="center">196</td>
</tr>
<tr>
<td align="center">MTTFA-LSTM</td>
<td align="center">305</td>
<td align="center">235</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">329</td>
<td align="center">264</td>
</tr>
<tr>
<td rowspan="4" align="center">80 Steps (20 h)</td>
<td align="center">CNN-GRU</td>
<td align="center">231</td>
<td align="center">123</td>
</tr>
<tr>
<td align="center">VMD-CNN-GRU</td>
<td align="center">297</td>
<td align="center">210</td>
</tr>
<tr>
<td align="center">MTTFA-LSTM</td>
<td align="center">330</td>
<td align="center">248</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">368</td>
<td align="center">277</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-4">
<title>4.4 Comparison with traditional models</title>
<p>In order to fully understand the difference between the forecasting performance of the DAM-GRU model and the traditional models, this section selects 4 traditional models including ARMA, RF, SVM and BP, and conducts comparative experiments using the same experimental strategy as in <xref ref-type="sec" rid="s4-3">Section 4.3</xref>. The autocorrelation order of ARMA is set to 25, and the moving average order is set to 3. The number of decision trees of RF model is 80, and the minimum number of leaves is 5. SVM adopts Radical Basis Function (RBF) as the kernel function to build a regression model with multidimensional variables. The BP neural network has 6 units in the input layer, which are used to input 5 features and the forecasting results from the previous step, the hidden layer has 15 units for high dimensional mapping, and the output layer has one node for multi-step cyclic forecasting. <xref ref-type="fig" rid="F8">Figure 8</xref> shows a comparison of the forecast curves and box plots for each model when forecasting 20, 40 and 80 steps.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>5-h, 10-h and 20-h forecast curves and box plots for 5 models including ARMA, RF, SVM, BP, and DAM-GRU.</p>
</caption>
<graphic xlink:href="fenrg-12-1346000-g008.tif"/>
</fig>
<p>From <xref ref-type="fig" rid="F8">Figure 8</xref> it can be seen that the ARMA model is slightly stable at 5 h, but the error increases significantly as the prediction time increases, which may be less suitable for long-term forecasting. The three models, RF, SVM and BP, also show significant lags and do not fit the observed curves well at the peaks, where power changes more frequently. The DAM-GRU model proposed in this paper is able to accurately capture most of the large power variations and achieve accurate forecasts. This shows that the ideas and time-series feature processing methods specifically designed for wind power forecasting can make good use of historical data and capture the patterns embedded in longer time periods. Combining <xref ref-type="table" rid="T6">Tables 6</xref>, <xref ref-type="table" rid="T7">7</xref>, it can be seen that although the proposed model has about twice the training and prediction time, its accuracy has improved significantly. As hardware performance improves, the training time for the model will be reduced even further in the future.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Comparison of multi-stage forecasting errors with traditional models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Forecasting horizon</th>
<th rowspan="2" align="center">Model</th>
<th colspan="6" align="center">Evaluation metrics</th>
</tr>
<tr>
<th align="center">RMSE</th>
<th align="center">MAE</th>
<th align="center">
<italic>R</italic>
<sup>2</sup>
</th>
<th align="center">SSE</th>
<th align="center">MAPE/%</th>
<th align="center">RRMSE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="5" align="center">20 Steps (5 h)</td>
<td align="center">ARMA</td>
<td align="center">2.4292</td>
<td align="center">1.9001</td>
<td align="center">0.9214</td>
<td align="center">118.0191</td>
<td align="center">24.7752</td>
<td align="center">0.3015</td>
</tr>
<tr>
<td align="center">RF</td>
<td align="center">2.4164</td>
<td align="center">1.8993</td>
<td align="center">0.9223</td>
<td align="center">116.7826</td>
<td align="center">23.0981</td>
<td align="center">0.2986</td>
</tr>
<tr>
<td align="center">SVM</td>
<td align="center">2.1266</td>
<td align="center">1.8459</td>
<td align="center">0.9398</td>
<td align="center">90.4450</td>
<td align="center">22.1179</td>
<td align="center">0.2698</td>
</tr>
<tr>
<td align="center">BP</td>
<td align="center">1.9926</td>
<td align="center">1.6608</td>
<td align="center">0.9471</td>
<td align="center">79.4128</td>
<td align="center">19.1813</td>
<td align="center">0.2324</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">0.6809</td>
<td align="center">0.4866</td>
<td align="center">0.9939</td>
<td align="center">9.2734</td>
<td align="center">5.5039</td>
<td align="center">0.0774</td>
</tr>
<tr>
<td rowspan="5" align="center">40 Steps (10 h)</td>
<td align="center">ARMA</td>
<td align="center">3.4827</td>
<td align="center">2.9137</td>
<td align="center">0.8855</td>
<td align="center">485.1629</td>
<td align="center">29.7151</td>
<td align="center">0.4426</td>
</tr>
<tr>
<td align="center">RF</td>
<td align="center">3.2436</td>
<td align="center">2.6804</td>
<td align="center">0.8871</td>
<td align="center">420.8490</td>
<td align="center">24.0143</td>
<td align="center">0.3536</td>
</tr>
<tr>
<td align="center">SVM</td>
<td align="center">3.0249</td>
<td align="center">2.6165</td>
<td align="center">0.9136</td>
<td align="center">366.0060</td>
<td align="center">23.7110</td>
<td align="center">0.3296</td>
</tr>
<tr>
<td align="center">BP</td>
<td align="center">2.5323</td>
<td align="center">2.0402</td>
<td align="center">0.9395</td>
<td align="center">256.4942</td>
<td align="center">17.2039</td>
<td align="center">0.2317</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">0.8822</td>
<td align="center">0.6644</td>
<td align="center">0.9926</td>
<td align="center">31.1283</td>
<td align="center">6.7024</td>
<td align="center">0.0856</td>
</tr>
<tr>
<td rowspan="5" align="center">80 Steps (20 h)</td>
<td align="center">ARMA</td>
<td align="center">4.5361</td>
<td align="center">3.5865</td>
<td align="center">0.7828</td>
<td align="center">1642.1259</td>
<td align="center">19.7657</td>
<td align="center">0.2577</td>
</tr>
<tr>
<td align="center">RF</td>
<td align="center">4.2578</td>
<td align="center">3.3195</td>
<td align="center">0.8086</td>
<td align="center">1450.3125</td>
<td align="center">19.1700</td>
<td align="center">0.2493</td>
</tr>
<tr>
<td align="center">SVM</td>
<td align="center">3.7247</td>
<td align="center">2.9615</td>
<td align="center">0.8536</td>
<td align="center">1109.8757</td>
<td align="center">16.7168</td>
<td align="center">0.2105</td>
</tr>
<tr>
<td align="center">BP</td>
<td align="center">3.3629</td>
<td align="center">2.7205</td>
<td align="center">0.8806</td>
<td align="center">904.7498</td>
<td align="center">15.6823</td>
<td align="center">0.1988</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">1.2521</td>
<td align="center">1.0003</td>
<td align="center">0.9835</td>
<td align="center">125.4133</td>
<td align="center">7.5511</td>
<td align="center">0.1062</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Comparison of training time and forecasting time with traditional models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Forecasting horizon</th>
<th align="center">Model</th>
<th align="center">Training time (s)</th>
<th align="center">Forecast time (ms)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="5" align="center">20 Steps (5 h)</td>
<td align="center">ARMA</td>
<td align="center">74</td>
<td align="center">14</td>
</tr>
<tr>
<td align="center">RF</td>
<td align="center">136</td>
<td align="center">26</td>
</tr>
<tr>
<td align="center">SVM</td>
<td align="center">138</td>
<td align="center">34</td>
</tr>
<tr>
<td align="center">BP</td>
<td align="center">129</td>
<td align="center">32</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">304</td>
<td align="center">262</td>
</tr>
<tr>
<td rowspan="5" align="center">40 Steps (10 h)</td>
<td align="center">ARMA</td>
<td align="center">81</td>
<td align="center">15</td>
</tr>
<tr>
<td align="center">RF</td>
<td align="center">155</td>
<td align="center">29</td>
</tr>
<tr>
<td align="center">SVM</td>
<td align="center">142</td>
<td align="center">30</td>
</tr>
<tr>
<td align="center">BP</td>
<td align="center">136</td>
<td align="center">37</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">329</td>
<td align="center">264</td>
</tr>
<tr>
<td rowspan="5" align="center">80 Steps (20 h)</td>
<td align="center">ARMA</td>
<td align="center">96</td>
<td align="center">19</td>
</tr>
<tr>
<td align="center">RF</td>
<td align="center">167</td>
<td align="center">31</td>
</tr>
<tr>
<td align="center">SVM</td>
<td align="center">171</td>
<td align="center">38</td>
</tr>
<tr>
<td align="center">BP</td>
<td align="center">181</td>
<td align="center">43</td>
</tr>
<tr>
<td align="center">DAM-GRU</td>
<td align="center">368</td>
<td align="center">277</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>Highly precise wind power forecasting is essential. Specifically, from the energy and environmental perspectives, it is conducive to the efficient utilization of wind power and the reduction of global carbon emissions; from the grid perspective, it is beneficial to the operator&#x2019;s response to the fluctuation of wind power and the rational allocation of power. In order to achieve this goal through both real-time feature selection and time complexity reduction, we propose a DAM-GRU model and derive the following conclusions:<list list-type="simple">
<list-item>
<p>1. The results of the comparative experiments show that the introduction of the FAM in this study effectively extracts features contributing significantly to the point being forecast, thus enhancing short-term wind power forecasting accuracy.</p>
</list-item>
<list-item>
<p>2. One-dimensional convolutions with different kernel sizes provide filtering effects, reducing the complexity of the wind power sequences in individual channels and making forecasts less challenging for the model.</p>
</list-item>
<list-item>
<p>3. The temporal attention mechanism extracts crucial temporal features of preliminary forecasts at different time steps, while the addition of MHTAM helps the GRU network extract significant temporal features from multiple channels.</p>
</list-item>
</list>
</p>
<p>The proposed DAM-GRU algorithm is investigated with measured wind power data from a wind farm in Inner Mongolia, and better forecasting results are obtained, which will provide an effective basis for the construction of new wind farms in the neighbourhood. For example, accurate wind power forecasting will reduce the pressure on nearby peak frequency regulation power plants and increase the installed capacity of wind power generation, as well as facilitate the measurement of operating costs and the development of maintenance plans for the newly built wind power plants. However, this study also has the shortcoming that the setting of hyperparameters is based on the tuning experiment, which may cause the model to fall into local optimisation. In addition, the study is insufficient for the generalisation ability of the model. In the future, we will take the optimisation algorithm and the generalisation ability test as a breakthrough point to improve the model.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation. Requests to access these datasets should be directed to YL, <email>21214037870025@ymu.edu.cn</email>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>WX: Funding acquisition, Software, Writing&#x2013;review and editing. YL: Methodology, Writing&#x2013;original draft. XF: Validation, Writing&#x2013;review and editing. ZS: Writing&#x2013;original draft. QW: Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work is supported by the National Natural Science Foundation of China (U1802271), and by the Ethnic and Religious Affairs Commission of Yunnan Province (2023YNMW010).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdoos</surname>
<given-names>A. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A new intelligent method based on combination of vmd and elm for short term wind power forecasting</article-title>. <source>Neurocomputing</source> <volume>203</volume>, <fpage>111</fpage>&#x2013;<lpage>120</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2016.03.054</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Karasu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zio</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A new hybrid model for wind speed forecasting combining long short-term memory neural network, decomposition methods and grey wolf optimizer</article-title>. <source>Appl. Soft Comput.</source> <volume>100</volume>, <fpage>106996</fpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2020.106996</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chi</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Wind power prediction based on wt-bigru-attention-tcn model</article-title>. <source>Front. Energy Res.</source> <volume>11</volume>, <fpage>1156007</fpage>. <pub-id pub-id-type="doi">10.3389/fenrg.2023.1156007</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Couto</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Estanqueiro</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Enhancing wind power forecast accuracy using the weather research and forecasting numerical model-based features and artificial neuronal networks</article-title>. <source>Renew. Energy</source> <volume>201</volume>, <fpage>1076</fpage>&#x2013;<lpage>1085</lpage>. <pub-id pub-id-type="doi">10.1016/j.renene.2022.11.022</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Short-term wind power forecasting using the hybrid model of improved variational mode decomposition and correntropy long short-term memory neural network</article-title>. <source>Energy</source> <volume>214</volume>, <fpage>118980</fpage>. <pub-id pub-id-type="doi">10.1016/j.energy.2020.118980</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Farah</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>David A</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Humaira</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Aneela</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Steffen</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Short-term multi-hour ahead country-wide wind power prediction for Germany using gated recurrent unit deep learning</article-title>. <source>Renew. Sustain. Energy Rev.</source> <volume>167</volume>, <fpage>112700</fpage>. <pub-id pub-id-type="doi">10.1016/j.rser.2022.112700</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A multichannel-based cnn and gru method for short-term wind power prediction</article-title>. <source>Electronics</source> <volume>12</volume>, <fpage>4479</fpage>. <pub-id pub-id-type="doi">10.3390/electronics12214479</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Giebel</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kariniotakis</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Wind power forecasting&#x2014;a review of the state of the art</article-title>. <source>Renew. energy Forecast.</source>, <fpage>59</fpage>&#x2013;<lpage>109</lpage>. <pub-id pub-id-type="doi">10.1016/b978-0-08-100504-0.00003-2</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Short-term wind power prediction based on eemd&#x2013;lasso&#x2013;qrnn model</article-title>. <source>Appl. Soft Comput.</source> <volume>105</volume>, <fpage>107288</fpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2021.107288</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Wind power forecasting using attention-based recurrent neural networks: a comparative study</article-title>. <source>IEEE Access</source> <volume>9</volume>, <fpage>40432</fpage>&#x2013;<lpage>40444</lpage>. <pub-id pub-id-type="doi">10.1109/access.2021.3065502</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A short-term wind power forecasting model based on 3d convolutional neural network&#x2013;gated recurrent unit</article-title>. <source>Sustainability</source> <volume>15</volume>, <fpage>14171</fpage>. <pub-id pub-id-type="doi">10.3390/su151914171</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ju</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Research on short-term wind power prediction of gru based on similar days</article-title>. <source>J. Phys.: Conf. Ser.</source> <volume>2087</volume>, <fpage>012089</fpage>. <pub-id pub-id-type="doi">10.1088/1742-6596/2087/1/012089</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A novel hybrid methodology for short-term wind power forecasting based on adaptive neuro-fuzzy inference system</article-title>. <source>Renew. energy</source> <volume>103</volume>, <fpage>620</fpage>&#x2013;<lpage>629</lpage>. <pub-id pub-id-type="doi">10.1016/j.renene.2016.10.074</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Ultra-short-term wind power forecasting based on deep bayesian model with uncertainty</article-title>. <source>Renew. Energy</source> <volume>205</volume>, <fpage>598</fpage>&#x2013;<lpage>607</lpage>. <pub-id pub-id-type="doi">10.1016/j.renene.2023.01.038</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Short-term multi-step ahead wind power predictions based on a novel deep convolutional recurrent network method</article-title>. <source>IEEE Trans. Sustain. Energy</source> <volume>12</volume>, <fpage>1820</fpage>&#x2013;<lpage>1833</lpage>. <pub-id pub-id-type="doi">10.1109/tste.2021.3067436</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The attention-assisted ordinary differential equation networks for short-term probabilistic wind power predictions</article-title>. <source>Appl. Energy</source> <volume>324</volume>, <fpage>119794</fpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2022.119794</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Short-term wind power forecasting based on multivariate/multi-step lstm with temporal feature attention mechanism</article-title>. <source>Appl. Soft Comput.</source> <volume>150</volume>, <fpage>111050</fpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2023.111050</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meng</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>A hybrid deep learning architecture for wind power prediction based on bi-attention mechanism and crisscross optimization</article-title>. <source>Energy</source> <volume>238</volume>, <fpage>121795</fpage>. <pub-id pub-id-type="doi">10.1016/j.energy.2021.121795</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Saini</surname>
<given-names>V. K.</given-names>
</name>
<name>
<surname>Bhardwaj</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Mathur</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Gated recurrent unit (gru) based short term forecasting for wind energy estimation</article-title>,&#x201d; in <conf-name>2020 International Conference on Power, Energy, Control and Transmission Systems (ICPECTS)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Santhosh</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Venkaiah</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>D. V.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Ensemble empirical mode decomposition based adaptive wavelet neural network method for wind speed prediction</article-title>. <source>Energy Convers. Manag.</source> <volume>168</volume>, <fpage>482</fpage>&#x2013;<lpage>493</lpage>. <pub-id pub-id-type="doi">10.1016/j.enconman.2018.04.099</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shih</surname>
<given-names>S.-Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>F.-K.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>H.-y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Temporal pattern attention for multivariate time series forecasting</article-title>. <source>Mach. Learn.</source> <volume>108</volume>, <fpage>1421</fpage>&#x2013;<lpage>1441</lpage>. <pub-id pub-id-type="doi">10.1007/s10994-019-05815-0</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Design of a wind power forecasting system based on deep learning</article-title>. <source>J. Phys.: Conf. Ser.</source> <volume>2562</volume>, <fpage>012043</fpage>. <pub-id pub-id-type="doi">10.1088/1742-6596/2562/1/012043</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Short-term wind power forecasting based on vmd decomposition, convlstm networks and error analysis</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>134422</fpage>&#x2013;<lpage>134434</lpage>. <pub-id pub-id-type="doi">10.1109/access.2020.3011060</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Backtracking search optimization algorithm-based least square support vector machine and its applications</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>94</volume>, <fpage>103801</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2020.103801</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Modes decomposition forecasting approach for ultra-short-term wind speed</article-title>. <source>Appl. Soft Comput.</source> <volume>105</volume>, <fpage>107303</fpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2021.107303</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021a</year>). <article-title>Multi-step short-term wind speed prediction based on integrated multi-model fusion</article-title>. <source>Appl. Energy</source> <volume>298</volume>, <fpage>117248</fpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2021.117248</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021b</year>). <article-title>A novel decomposition-ensemble prediction model for ultra-short-term wind speed</article-title>. <source>Energy Convers. Manag.</source> <volume>248</volume>, <fpage>114775</fpage>. <pub-id pub-id-type="doi">10.1016/j.enconman.2021.114775</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A combination forecasting model of wind speed based on decomposition</article-title>. <source>Energy Rep.</source> <volume>7</volume>, <fpage>1217</fpage>&#x2013;<lpage>1233</lpage>. <pub-id pub-id-type="doi">10.1016/j.egyr.2021.02.002</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A prediction approach using ensemble empirical mode decomposition-permutation entropy and regularized extreme learning machine for short-term wind speed</article-title>. <source>Wind Energy</source> <volume>23</volume>, <fpage>177</fpage>&#x2013;<lpage>206</lpage>. <pub-id pub-id-type="doi">10.1002/we.2422</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>van Heerden</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Vermeulen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>van Staden</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Wind power forecasting using hybrid recurrent neural networks with empirical mode decomposition</article-title>,&#x201d; in <conf-name>2022 IEEE International Conference on Environment and Electrical Engineering and 2022 IEEE Industrial and Commercial Power Systems Europe (EEEIC/I&#x26;CPS Europe)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A deep asymmetric laplace neural network for deterministic and probabilistic wind power forecasting</article-title>. <source>Renew. Energy</source> <volume>196</volume>, <fpage>497</fpage>&#x2013;<lpage>517</lpage>. <pub-id pub-id-type="doi">10.1016/j.renene.2022.07.009</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A review of wind speed and wind power forecasting with deep neural networks</article-title>. <source>Appl. Energy</source> <volume>304</volume>, <fpage>117766</fpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2021.117766</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Boosted gru model for short-term forecasting of wind power with feature-weighted principal component analysis</article-title>. <source>Energy</source> <volume>267</volume>, <fpage>126503</fpage>. <pub-id pub-id-type="doi">10.1016/j.energy.2022.126503</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Tu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>D. F.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chao</surname>
<given-names>L. S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Modeling localness for self-attention networks</source>. <comment>
<italic>arXiv preprint arXiv:1810.10182</italic>
</comment>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A deep attention convolutional recurrent network assisted by k-shape clustering and enhanced memory for short term wind speed predictions</article-title>. <source>IEEE Trans. Sustain. Energy</source> <volume>13</volume>, <fpage>856</fpage>&#x2013;<lpage>867</lpage>. <pub-id pub-id-type="doi">10.1109/tste.2021.3135278</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>An improved mixture density network via wasserstein distance based adversarial learning for probabilistic wind speed predictions</article-title>. <source>IEEE Trans. Sustain. Energy</source> <volume>13</volume>, <fpage>755</fpage>&#x2013;<lpage>766</lpage>. <pub-id pub-id-type="doi">10.1109/tste.2021.3131522</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Multi-source and temporal attention network for probabilistic wind power prediction</article-title>. <source>IEEE Trans. Sustain. Energy</source> <volume>12</volume>, <fpage>2205</fpage>&#x2013;<lpage>2218</lpage>. <pub-id pub-id-type="doi">10.1109/tste.2021.3086851</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Short-term wind speed prediction model based on ga-ann improved by vmd</article-title>. <source>Renew. Energy</source> <volume>156</volume>, <fpage>1373</fpage>&#x2013;<lpage>1388</lpage>. <pub-id pub-id-type="doi">10.1016/j.renene.2019.12.047</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Hybrid vmd-cnn-gru-based model for short-term forecasting of wind power considering spatio-temporal features</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>121</volume>, <fpage>105982</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2023.105982</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>