<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Energy Res.</journal-id>
<journal-title>Frontiers in Energy Research</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Energy Res.</abbrev-journal-title>
<issn pub-type="epub">2296-598X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1501963</article-id>
<article-id pub-id-type="doi">10.3389/fenrg.2024.1501963</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Energy Research</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>An optimized method for short-term load forecasting based on feature fusion and ConvLSTM-3D neural network</article-title>
<alt-title alt-title-type="left-running-head">Yang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenrg.2024.1501963">10.3389/fenrg.2024.1501963</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Xiaofeng</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Shousheng</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Li</surname>
<given-names>Kangyi</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2853139/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Wenjin</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Si</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Jingwei</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff>
<institution>State Grid Zhejiang Electric Power Co., Ltd Shaoxing Power Supply Company</institution>, <addr-line>Shaoxing</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2365868/overview">Pasquale De Falco</ext-link>, University of Naples Parthenope, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1517415/overview">Lipeng Zhu</ext-link>, Hunan University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2580189/overview">Antonio Bracale</ext-link>, University of Naples Parthenope, Italy</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Kangyi Li, <email>ky_li135@163.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>01</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1501963</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>09</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>16</day>
<month>12</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Yang, Zhao, Li, Chen, Zhang and Chen.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Yang, Zhao, Li, Chen, Zhang and Chen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>As renewable energy continues to penetrate modern power systems, accurate short-term load forecasting is crucial for optimizing power generation resource allocation and reducing operational costs. Traditional forecasting methods often overlook key factors such as holiday load variations and differences in user electricity consumption behavior, resulting in reduced accuracy. To address this, we propose an optimized short-term load forecasting method based on time and weather-fused features using a ConvLSTM-3D neural network. The Prophet algorithm is first employed to decompose historical electricity load data, extracting feature components related to time variables. Simultaneously, the SHAP algorithm filters weather variables to identify highly correlated weather features. A time attention mechanism is then applied to fuse these features based on their correlation weights, enhancing their impact within the time series. Finally, the ConvLSTM-3D model is trained on the fused features to generate short-term load forecasts. A case study using real-world data validates the proposed method, demonstrating significant improvements in forecasting accuracy.</p>
</abstract>
<kwd-group>
<kwd>short-term load forecasting</kwd>
<kwd>fused features</kwd>
<kwd>prophet algorithm</kwd>
<kwd>SHAP algorithm</kwd>
<kwd>convlstm-3D model</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Smart Grids</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Accurate electricity load forecasting, particularly short-term electricity forecasting (STLF), is a vital component of the safe operation of contemporary power systems and provides significant guidance for energy dispatch (<xref ref-type="bibr" rid="B24">Yang et al., 2022</xref>; <xref ref-type="bibr" rid="B19">Si et al., 2023</xref>). However, on the electricity load demand side, with technological advancements and changing climate conditions, the usage of new electric equipment such as electric vehicles (EVs), air conditioners, and smart home devices has increased sharply (<xref ref-type="bibr" rid="B2">Ahmad et al., 2022</xref>; <xref ref-type="bibr" rid="B18">Pijarski and Belowski, 2024</xref>). This surge in usage introduces greater uncertainty into electrical load forecasts. Furthermore, electricity load is not only affected by weather conditions such as temperature, humidity, and precipitation, but also influenced by residential consumption habits at different times, such as holidays and weekly seasonality (<xref ref-type="bibr" rid="B24">Yang et al., 2022</xref>). Accurate load forecasting is an effective method for managing energy consumption, facilitating the production of a reliable electricity trade market, and ensuring the frequency safety and stability of the new power system (<xref ref-type="bibr" rid="B2">Ahmad et al., 2022</xref>). More importantly, as the complexity of power systems increases, the need for precise forecasts becomes crucial (<xref ref-type="bibr" rid="B1">Abdolrasol et al., 2021</xref>). Therefore, it is imperative to improve load forecasting accuracy, prompting more researchers to engage in this area of study.</p>
<p>Early load forecasting methods were primarily based on statistical models, such as ARMA, ARIMA, and HAR. While these statistical models have fewer parameters and offer relatively high computational efficiency, they struggle to process nonlinear data, making it difficult to meet the current demands of variable power load forecasting (<xref ref-type="bibr" rid="B2">Ahmad et al., 2022</xref>). With the advancement of artificial intelligence and the increased computing power of modern systems, load forecasting methods based on artificial neural networks (ANNs) have gained widespread use (<xref ref-type="bibr" rid="B1">Abdolrasol et al., 2021</xref>). Mainstream ANNs used for power system load forecasting include models like SVR, GRU, LSTM, and BP neural networks. Compared to statistical methods, ANN-based models can effectively capture the temporal characteristics of nonlinear data, leading to more accurate predictions (<xref ref-type="bibr" rid="B15">Liu et al., 2018</xref>; <xref ref-type="bibr" rid="B20">Wang et al., 2012</xref>). However, the performance of these models is highly sensitive to the choice of structural parameters and the quality of the training data. This can result in issues such as overfitting or underfitting. Additionally, these models require data sets with high integrity and appropriate sampling rates (<xref ref-type="bibr" rid="B23">Wazirali et al., 2023</xref>). The large variations in load data collected across different devices and regions also place high demands on the robustness of prediction models. Consequently, despite their advantages, ANN-based models may struggle to meet the evolving needs of today&#x2019;s load forecasting (<xref ref-type="bibr" rid="B9">Ding et al., 2015</xref>).</p>
<p>With the continued advancement of artificial intelligence, load forecasting methods based on deep learning models have been widely adopted in recent years. Compared to traditional artificial neural networks, deep learning models feature more hidden layers and possess powerful feature extraction capabilities. For example, reference (<xref ref-type="bibr" rid="B10">Dong et al., 2024</xref>) proposes a multi-node load forecasting method for power systems using a deep learning multi-time scale convolutional model integrated with a Transformer model. This fused model excels at capturing the temporal and spatial characteristics of data, resulting in notable improvements in multi-node power system load forecasting. Similarly, reference (<xref ref-type="bibr" rid="B28">Zhuang et al., 2023</xref>) leverages a graph attention network and a one-dimensional convolutional neural network to extract the temporal and spatial characteristics of regional loads, allowing the model to better explore the spatial dependencies of regional loads and provide richer feature variables for prediction. Reference (<xref ref-type="bibr" rid="B13">Huang et al., 2023</xref>) introduces a deep learning model combining Spearman correlation, GCN, and GRU. The Spearman correlation coefficient quantifies the relationship between loads at different nodes, while the GCN captures spatial correlations, and GRU mines the temporal relationships within the data. A well-designed deep learning model can fully explore the latent features in data, playing a critical role in enhancing prediction accuracy. Given the complexity of feature variables involved in short-term load forecasting&#x2014;particularly for bus loads&#x2014;the model employed in this paper must have strong feature extraction capabilities.</p>
<p>In short-term load forecasting, the selection of training features is crucial to achieving accurate predictions. Current time-series-based prediction methods typically utilize historical data, meteorological data, and calendar data for forecasting (<xref ref-type="bibr" rid="B14">Liu et al., 2020</xref>). However, the influence and weight of different meteorological and calendar data on load forecasting vary significantly. Using these data directly as features can result in suboptimal predictions, and an excessive number of features can introduce redundancy. Furthermore, calendar data generally only include broad information such as holidays, weekdays, and workdays, which fails to capture the specific effects of various time-related factors on load (<xref ref-type="bibr" rid="B8">Dahl et al., 2018</xref>). To address these limitations, reference (<xref ref-type="bibr" rid="B25">Yang et al., 2023</xref>) employs multidimensional time domain features as model inputs for forecasting. First, a load feature decomposition model based on a periodic trend decomposition algorithm is constructed to obtain feature components that reflect the load&#x2019;s trend, periodicity, and randomness. Similarly, reference (<xref ref-type="bibr" rid="B27">Zhan et al., 2022</xref>) decomposes load data into a series of sub-modal data with different central frequencies through variational mode decomposition, and clusters these sub-modal data to extract feature quantities of different frequency centers. Reference (<xref ref-type="bibr" rid="B6">Chen C. et al., 2024</xref>) introduces derivative terms, using the difference between vector values as supplementary features to capture load change characteristics across different time periods. While these methods have achieved certain improvements by optimizing feature selection, they do not comprehensively consider the influence weights of specific features, such as time and weather features. Thus, further optimization of feature processing is still needed to enhance the accuracy of short-term load forecasting.</p>
<p>Building on these approaches, this paper proposes a short-term load forecasting method that integrates time and weather fusion features with a ConvLSTM-3D neural network. First, the historical power load data is preprocessed to align with corresponding timestamps and labeled with relevant data such as holidays, years, seasons, months, weeks, workdays, and non-workdays. Next, the Prophet algorithm is employed to extract time feature components from the dataset, generating various feature quantities related to power load. Concurrently, the SHAP algorithm is utilized to filter weather variables, identifying the most strongly correlated weather feature components. Based on the correlation weights of these feature components, a time attention mechanism is applied to fuse the features, effectively combining the time and weather components while increasing their respective influence on the time series. Finally, the fused features are input into the ConvLSTM-3D model to train the system and produce future short-term power load forecasts.</p>
</sec>
<sec id="s2">
<title>2 Time-weather fusion features</title>
<sec id="s2-1">
<title>2.1 Time component extraction using the prophet algorithm</title>
<p>Compared to ultra-short-term load forecasting, short-term power load forecasting operates on a relatively longer time scale. As a result, power load is influenced by various time-related factors, such as users&#x2019; power consumption habits, holiday patterns, and power demand across different periods. Additionally, the impact and significance of these time components on power consumption vary. Therefore, to enhance the accuracy of short-term power load forecasting, it is essential to incorporate diverse time components as key features in the prediction model.</p>
<p>The Prophet algorithm, developed by Facebook, is a time series forecasting tool designed to handle data with strong seasonality and is highly robust to missing data and sudden trend changes (<xref ref-type="bibr" rid="B5">Ceperic et al., 2013</xref>). Compared to traditional algorithms, Prophet offers simpler parameter tuning and allows users to adapt parameters to different scenarios. In this paper, the Prophet algorithm is employed to extract various time components from the original power load data, including trend components, holiday characteristics, weekly characteristics, and daily characteristics. The calculation expression for this process is shown in <xref ref-type="disp-formula" rid="e1">Equation 1</xref>.<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the original power load, <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the trend term, which represents the trend of the time series in the non-periodic aspect, <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the periodic term; <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the holiday term, it represents the impact of the potential non-fixed periodic holidays in the time series on the predicted value, <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the error term, which represents the fluctuation predicted by the model.</p>
<p>In the trend term, two important functions are employed: one based on logistic regression (non-linear growth) and the other based on piecewise linear functions (linear growth). The trend component base4d on logistic regression can be expressed as <xref ref-type="disp-formula" rid="e2">Equation 2</xref>.<disp-formula id="e2">
<mml:math id="m7">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>a</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>a</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the carrying capacity, which is a function that changes with time and limits the maximum value that can grow. <inline-formula id="inf7">
<mml:math id="m9">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the growth rate, and <inline-formula id="inf8">
<mml:math id="m10">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the offset. In addition, the trend term expression based on the piecewise linear function is in <xref ref-type="disp-formula" rid="e3">Equation 3</xref>.<disp-formula id="e3">
<mml:math id="m11">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi mathvariant="bold-italic">&#x3b4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi mathvariant="bold-italic">&#x3b3;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf9">
<mml:math id="m12">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the growth rate, <inline-formula id="inf10">
<mml:math id="m13">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the change in the growth rate, and <inline-formula id="inf11">
<mml:math id="m14">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the offset. Since the short-term forecast data of the load belongs to nonlinear growth, the growth term is represented by logistic regression. For the periodic term, the Prophet algorithm uses Fourier series to represent it, and its expression is <xref ref-type="disp-formula" rid="e4">Equation 4</xref>.<disp-formula id="e4">
<mml:math id="m15">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m16">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>In the formula, <inline-formula id="inf12">
<mml:math id="m17">
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the period in days; <inline-formula id="inf13">
<mml:math id="m18">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the set of smoothing coefficients <inline-formula id="inf14">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf15">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which satisfies the normal distribution; N is the number of smoothing coefficients <inline-formula id="inf16">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> or <inline-formula id="inf17">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf18">
<mml:math id="m23">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the sequence number of the smoothing coefficient. The holiday term can be represented as <xref ref-type="disp-formula" rid="e6">Equation 6</xref>.<disp-formula id="e6">
<mml:math id="m24">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>&#x3ba;</mml:mi>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf19">
<mml:math id="m25">
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the regression matrix, and <inline-formula id="inf20">
<mml:math id="m26">
<mml:mrow>
<mml:mi>&#x3ba;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the prior change parameter corresponding to holidays.</p>
<p>On the other hand, the hyperparameters of the Prophet model include the changepoint prior scale (CPS), seasonality prior scale (SPS), and holiday prior scale (HPS). The CPS dictates the model&#x2019;s sensitivity to trend shifts, SPS regulates the strength of the seasonal components, and HPS governs the magnitude of holiday effects. The training and optimization process can be summarized as follows.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>The training and optimization process of Prophet algorithm.<list list-type="simple">
<list-item>
<p>
<bold>Define</bold> parameter ranges</p>
</list-item>
<list-item>
<p>&#x2003;CPS from 0.001 to 0.5</p>
</list-item>
<list-item>
<p>&#x2003;SPS from 0.01 to 10</p>
</list-item>
<list-item>
<p>&#x2003;HPS from 0.01 to 10</p>
</list-item>
<list-item>
<p>
<bold>Initialize:</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;best_params as empty dictionary</p>
</list-item>
<list-item>
<p>&#x2003;best_perf as infinity</p>
</list-item>
<list-item>
<p>
<bold>For</bold> each value <bold>in</bold> CPS</p>
</list-item>
<list-item>
<p>&#x2003;<bold>For</bold> each value <bold>in</bold> SPS</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<bold>For</bold> each value <bold>in</bold> PHS</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<bold>Set</bold> model <bold>parameters</bold> (cps, sps, hps)</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<bold>Fit</bold> model on training data</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;Evaluate model on validation data using <bold>RMSE</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;<bold>If</bold> current RMSE &#x3c; best_performance</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<bold>Update</bold> best_performance to current RMSE</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;<bold>Update</bold> best_parameters to (CPS, SPS, HPS)</p>
</list-item>
<list-item>
<p>
<bold>Output</bold> best_params and best_perf</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
<sec id="s2-2">
<title>2.2 Feature selection using SHAP algorithm</title>
<p>Different feature quantities have varying influence weights on the load. To enhance the interpretability of these feature components within the prediction model, this paper employs the SHAP (Shapley Additive Explanations) model to explain the contribution of each feature (<xref ref-type="bibr" rid="B7">Chen W. et al., 2024</xref>). SHAP is rooted in game theory and not only measures feature contributions in individual predictions but also aggregates the overall explanation of the model for local results. By calculating the Shapley value for each feature, SHAP provides the average contribution of each feature to the model&#x2019;s predictions. The larger the SHAP value, the greater the influence of that feature on the model. Based on these values, features can be ranked, allowing us to identify which ones have the most significant impact on the prediction. For a load model, it can be characterized as <xref ref-type="disp-formula" rid="e7">Equation 7</xref>.<disp-formula id="e7">
<mml:math id="m27">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>f</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>The influence of feature <inline-formula id="inf21">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> on the output result is related to its corresponding coefficient. Accordingly, the incremental contribution of feature <inline-formula id="inf22">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is shown in <xref ref-type="disp-formula" rid="e8">Equation 8</xref>.<disp-formula id="e8">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3d5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>f</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>The Shapley value is calculated based on the average marginal contribution. For a given feature set <italic>S</italic>, the Shapley value for feature <italic>i</italic> is calculated using the following <xref ref-type="disp-formula" rid="e9">Equation 9</xref>:<disp-formula id="e9">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3d5;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x2286;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo>&#x7c;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>!</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>!</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>!</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xb7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x222a;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>where <inline-formula id="inf23">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3d5;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the SHAP value, <inline-formula id="inf24">
<mml:math id="m33">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the predicted output when the model only uses the feature set <italic>S</italic>, <inline-formula id="inf25">
<mml:math id="m34">
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is the number of features in the set <italic>S</italic>, <inline-formula id="inf26">
<mml:math id="m35">
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of all features, and <inline-formula id="inf27">
<mml:math id="m36">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x222a;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="&#x7c;">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the marginal contribution when feature <italic>i</italic> is included compared to when it is not including.</p>
</sec>
<sec id="s2-3">
<title>2.3 Feature reconstruction based on attention mechanism</title>
<p>Since different feature components have varying influences on the load across different time periods, using the original feature data for prediction may reduce the influence weight of certain features over time. To address this, a time attention mechanism is applied to reconstruct the fused features. This paper employs an attention mechanism based on the LSTM computing unit to reconstruct the time series with different feature components, optimizing their impact on the time scale. The structure of this approach is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Feature reconstruction model based on attention mechanism.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g001.tif"/>
</fig>
<p>Taking fusion features <inline-formula id="inf28">
<mml:math id="m37">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>L</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> as an example, the similarity score between the hidden state and different attributes in LSTM is calculated based on the hidden state <italic>h</italic>
<sub>t-1</sub> at time <italic>t</italic>-1. This score is then input into the Softmax function for normalization. The normalized similarity score is used to update the new feature sequence. After feature fusion, its expression can be represented as <xref ref-type="disp-formula" rid="e10">Equation 10</xref>.<disp-formula id="e10">
<mml:math id="m38">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>L</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>where <inline-formula id="inf29">
<mml:math id="m39">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the fusion feature with k dimensions, and <italic>L</italic> represents the time length of the feature component. The input component after feature encoding is expressed as <xref ref-type="disp-formula" rid="e11">Equation 11</xref>.<disp-formula id="e11">
<mml:math id="m40">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>&#x3b5;</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf30">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>&#x3b5;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf31">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>,and <inline-formula id="inf32">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are parameters obtained based on LSTM structure training. The weight coefficients corresponding to different feature components can be expressed as <xref ref-type="disp-formula" rid="e12">Equation 12</xref>.<disp-formula id="e12">
<mml:math id="m44">
<mml:mrow>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3be;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3be;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>The feature components reconstructed through the attention mechanism can be represented as <xref ref-type="disp-formula" rid="e13">Equation 13</xref>.<disp-formula id="e13">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mn>1</mml:mn>
</mml:msubsup>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
<mml:mn>1</mml:mn>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
</p>
<p>The combination of Prophet and SHAP uniquely addresses the limitations of traditional feature extraction and selection methods by leveraging Prophet&#x2019;s robust decomposition of complex time series patterns into trend, seasonality, and holiday effects, while SHAP provides precise, interpretable feature importance. This synergy ensures enhanced interpretability, accurate feature attribution, and the ability to manage complex, nonlinear interactions, resulting in more robust and explainable forecasting models. Furthermore, the fusion features are optimized through the attention mechanism, effectively refining the impact of various feature components over different time scales. The process of constructing weather features based on time and weather fusion used in this paper is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Feature reconstruction model based on attention mechanism.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g002.tif"/>
</fig>
</sec>
</sec>
<sec id="s3">
<title>3 Construction of short-term load forecasting model</title>
<p>The fusion features constructed using the above method incorporate both time feature components and important weather feature components, with all components represented as time series. As a result, the fusion feature can be viewed as a three-dimensional feature with a time series dimension (<xref ref-type="bibr" rid="B22">Wang et al., 2021</xref>). To effectively model this, the ConvLSTM-3D deep learning network, which uses ConvLSTM as its basic structural unit, is adopted as the prediction model. The short-term load forecasting model and process developed in this paper are illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<p>Compared to standard ConvLSTM, ConvLSTM-3D applies three-dimensional convolution operations in the input gate, forget gate, output gate, and cell state update, allowing for better handling of three-dimensional data (<xref ref-type="bibr" rid="B17">Moon et al., 2020</xref>; <xref ref-type="bibr" rid="B16">Mohammad et al., 2023</xref>). The input to each gate of the ConvLSTM unit contains three elements: memory information from the previous unit, output from the previous unit, and the input at the current time step. ConvLSTM consists of an input gate, forget gate, output gate, and memory unit. The model&#x2019;s network structure is depicted in <xref ref-type="fig" rid="F3">Figure 3A</xref>, while the schematic diagram of the internal structure of the unit is shown in <xref ref-type="fig" rid="F3">Figure 3B</xref> (<xref ref-type="bibr" rid="B11">Guo et al., 2021</xref>; <xref ref-type="bibr" rid="B3">Alhussein et al., 2020</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Forecasting model structure based on ConvLSTM. <bold>(A)</bold> ConvLSTM model structure diagram. <bold>(B)</bold> Internal structure of ConvLSTM unit.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g003.tif"/>
</fig>
<sec id="s3-1">
<title>3.1 Input gate</title>
<p>The input gate determines which parts of the current input should be updated in the memory cell. This process involves two steps: first, the sigmoid layer determines which information needs to be updated, and second, the tanh function generates candidate information. The structure of the input gate can be expressed as <xref ref-type="disp-formula" rid="e14">Equation 14</xref>.<disp-formula id="e14">
<mml:math id="m46">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>&#x3c7;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>&#x3c7;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>where <inline-formula id="inf33">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf34">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the output of the input gate and the backup information of the memory unit respectively, <inline-formula id="inf35">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the input data at the current time t, <inline-formula id="inf36">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the state information before the memory unit, <inline-formula id="inf37">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the output of the previous hidden layer unit of the ConvLSTM unit, <inline-formula id="inf38">
<mml:math id="m52">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf39">
<mml:math id="m53">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are the weight and bias of the input gate, respectively. <inline-formula id="inf40">
<mml:math id="m54">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the sigmoid activation function, <inline-formula id="inf41">
<mml:math id="m55">
<mml:mrow>
<mml:mi>tanh</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the hyperbolic tangent function, and <inline-formula id="inf42">
<mml:math id="m56">
<mml:mrow>
<mml:mo>&#x2217;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the convolution operation.</p>
</sec>
<sec id="s3-2">
<title>3.2 Forget gate</title>
<p>The forget gate selectively discards unnecessary information from the memory unit from the previous time step. The forget gate can be expressed as <xref ref-type="disp-formula" rid="e15">Equation 15</xref>.<disp-formula id="e15">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>&#x3c7;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>where <inline-formula id="inf43">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the weight of the information of the input layer flowing into the forget gate at this moment, <inline-formula id="inf44">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the weight of the final result of the previous hidden layer neural unit when the forget gate is input, <inline-formula id="inf45">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the weight of the memory unit state at the previous moment flowing into the forget gate, and <inline-formula id="inf46">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the bias parameter when the forget gate is calculated.</p>
</sec>
<sec id="s3-3">
<title>3.3 Memory cell</title>
<p>The current memory unit state information <inline-formula id="inf47">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is updated by describing the past long-term state and the current state. The process of updating the state information can be specifically expressed as <xref ref-type="disp-formula" rid="e16">Equation 16</xref>.<disp-formula id="e16">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-4">
<title>3.4 Output gate</title>
<p>The output gate also consists of two parts. One part is the information input <inline-formula id="inf48">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>,obtained by combining the short-term memory with the current input information (the output of the output gate at the current moment), and the other part <inline-formula id="inf49">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the final output after combining the long-term memory (the output of ConvLSTM at the current moment) which can be expressed as <xref ref-type="disp-formula" rid="e17">Equation 17</xref>.<disp-formula id="e17">
<mml:math id="m66">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>&#x3c7;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>tanh</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
</p>
<p>Compared with the traditional ConvLSTM model, this paper changes the loss function of the traditional ConvLSTM to make it more suitable for data features based on three-dimensional feature sequences. Its calculation is <xref ref-type="disp-formula" rid="e18">Equation 18</xref> (<xref ref-type="bibr" rid="B22">Wang et al., 2021</xref>):<disp-formula id="e18">
<mml:math id="m67">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>a</mml:mi>
</mml:msup>
<mml:mo>&#x2022;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:msup>
<mml:mo>&#x2022;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>where <inline-formula id="inf50">
<mml:math id="m68">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf51">
<mml:math id="m69">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf52">
<mml:math id="m70">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represent brightness similarity index, contrast similarity index and structure similarity index respectively. The structure of the prediction model is shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. The short-term load forecasting process based on fusion features and ConvLSTM-3D model is shown in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Prediction model structure based on ConvLSTM.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Short-term load forecasting model structure based on fusion features and deep learning.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4 Case analysis</title>
<sec id="s4-1">
<title>4.1 Data description and experimental simulation platform</title>
<p>To verify the effectiveness of the fusion features and deep learning models proposed in this paper for short-term load forecasting, relevant measured data from a power grid company in a region of Zhejiang is used. The load data covers the power consumption of certain areas in the region from 2021 to 2023, with a sampling rate of one data point every 15&#xa0;min. The weather data includes information such as surface temperature, wind speed, wind direction, and humidity from various areas within the region, also recorded at the same sampling rate. Key information about the load and meteorological data is summarized in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Important information of the measured data set.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center">Details</th>
<th align="center">Sampling Rate (min)</th>
<th align="center">Latitude and longitude</th>
<th align="center">Experimental Division</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="center">Load</td>
<td align="center">Region1</td>
<td align="center">15</td>
<td align="center">(30.035000, 120.579000)</td>
<td align="center">Prediction Performance Validation</td>
</tr>
<tr>
<td align="center">Region2</td>
<td align="center">15</td>
<td align="center">(29.695000, 120.840000)</td>
<td rowspan="2" align="center">Robustness Validation</td>
</tr>
<tr>
<td align="center">Region3</td>
<td align="center">15</td>
<td align="center">(29.748000, 120.385000)</td>
</tr>
<tr>
<td rowspan="6" align="center">Weather</td>
<td align="center">Surface Temperature</td>
<td align="center">15</td>
<td rowspan="6" align="center">---</td>
<td rowspan="6" align="center">Prediction and Robustness Experiments</td>
</tr>
<tr>
<td align="center">Wind Speed, Wind Direction</td>
<td align="center">15</td>
</tr>
<tr>
<td align="center">Humidity</td>
<td align="center">15</td>
</tr>
<tr>
<td align="center">Surface Pressure</td>
<td align="center">15</td>
</tr>
<tr>
<td align="center">Total Cloud Cover</td>
<td align="center">15</td>
</tr>
<tr>
<td align="center">Solar Irradiance</td>
<td align="center">15</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>For verification, this paper uses the total social load data of a city in the region, with 80% of the data used as the training set and the remaining 20% for the test set. Additionally, to verify the robustness of the proposed method, load data from three other regions is used as supplementary verification, with corresponding meteorological data from these regions included. To compare and assess the computational efficiency of different experimental methods, all experiments are conducted on a unified experimental platform. The software platform is built using Python-based TensorFlow and PyTorch, while the hardware platform consists of an Intel Core i7-11700 (CPU) and an NVIDIA GeForce GTX 1660 Ti (GPU).</p>
</sec>
<sec id="s4-2">
<title>4.2 Evaluation metrics</title>
<p>In order to verify the prediction effect of the fusion features and deep learning model proposed in this paper on short-term power load, this paper uses the root mean square error (RMSE), mean absolute percentage error (MAPE) and determination coefficient (R square, R2) as the evaluation of the prediction model. Among them, RMSE is used to measure the distance between the actual value and the predicted value. MAPE is used to measure the percentage of the difference between the actual value and the actual value. R2 is used to measure the degree of explanation of the model for data changes. Its range is 0&#x2013;1, and the larger the value, the better the model fits the sample (<xref ref-type="bibr" rid="B21">Wang et al., 2020</xref>; <xref ref-type="bibr" rid="B4">Bashir et al., 2022</xref>). The related calculation formulas are as <xref ref-type="disp-formula" rid="e19">Equations 19</xref>&#x2013;<xref ref-type="disp-formula" rid="e21">21</xref>.<disp-formula id="e19">
<mml:math id="m71">
<mml:mrow>
<mml:mtext>RMSE</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>
<disp-formula id="e20">
<mml:math id="m72">
<mml:mrow>
<mml:mtext>MAPE</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2217;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(20)</label>
</disp-formula>
<disp-formula id="e21">
<mml:math id="m73">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">R</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(21)</label>
</disp-formula>where <inline-formula id="inf53">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the actual observation value, <inline-formula id="inf54">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the predicted value, <inline-formula id="inf55">
<mml:math id="m76">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is the average of the actual observation values, and <inline-formula id="inf56">
<mml:math id="m77">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the total number of observation points.</p>
</sec>
<sec id="s4-3">
<title>4.3 Forecasting results analysis</title>
<sec id="s4-3-1">
<title>4.3.1 Temporal feature extraction results</title>
<p>In the experiment, the Prophet algorithm was first applied to extract various time components of the region, with the results shown in <xref ref-type="fig" rid="F6">Figure 6</xref>. And the error distribution of the Prophet algorithm for annual load forecasting is shown in <xref ref-type="fig" rid="F7">Figure 7</xref>. As seen in the figure, the trend, holiday, week, and day components all have a significant impact on the overall power load in the region. Based on these observations, the trend, holiday, week, and day components were selected as the key time characteristics for the region.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Temporal feature components extracted using the Prophet algorithm.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Error distribution of the Prophet algorithm for annual load forecasting.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g007.tif"/>
</fig>
<p>In addition, the SHAP algorithm was used to identify important weather feature components. SHAP quantifies the contribution of each feature to the forecast results, enabling the identification and explanation of weather features that significantly impact the model&#x2019;s predictions (<xref ref-type="bibr" rid="B26">Yang et al., 2024</xref>). This method not only enhances the transparency of the model but also provides a scientific basis for further meteorological research and decision-making. The results of the SHAP analysis are shown in <xref ref-type="fig" rid="F8">Figure 8</xref>.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Feature component contribution analysis based on SHAP values.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g008.tif"/>
</fig>
<p>As seen in the figure, temperature changes have the greatest impact on power load among the weather characteristics. Significant fluctuations in output occur as the temperature increases or decreases. It is worth noting that total cloud cover has a much smaller impact on the output compared to the other six characteristics. Due to its minimal contribution, it is difficult to quantify using the SHAP algorithm and thus is not displayed in the figure. To more intuitively assess the importance of different meteorological components on the load, the average absolute SHAP value of each weather characteristic is used to rank their importance. The results of this ranking are shown in <xref ref-type="fig" rid="F9">Figure 9</xref>.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Feature importance ranking based on mean absolute value SHAP.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g009.tif"/>
</fig>
<p>The weather features were determined and selected based on the SHAP threshold value. In this study, a SHAP threshold of 200 was applied to distinguish the main contributors to feature importance from features exhibiting a rapid decline in importance. Consequently, the selected weather features in this research include temperature, irradiance, and surface pressure.</p>
</sec>
<sec id="s4-3-2">
<title>4.3.2 Forecasting result analysis</title>
<p>To verify the superiority of the method proposed in this paper, an ablation experiment was conducted to compare different feature sets and prediction models. For feature selection, the comparison includes a single time feature component, a single weather feature component, and the fusion feature component recommended in this study. Regarding the prediction models, a comparative analysis was performed between LSTM, CNN-LSTM, ConvLSTM, and ConvLSTM-3D (the model used in this paper). Additionally, to determine whether different time periods capture the characteristics of Chinese holidays, power load data from May 1st to May 3rd (during National Day) was compared with power load data from non-holiday periods in the region to evaluate the prediction performance. The 72-hour-ahead prediction comparison results are illustrated in <xref ref-type="fig" rid="F10">Figure 10</xref>.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Comparison of forecasting results from different methods across different periods. <bold>(A)</bold> weekday load forecasting results. <bold>(B)</bold> holiday load forecasting results.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g010.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F10">Figure 10A</xref> shows the load forecast curve from April 1 to 3 April 2023, representing the forecast results for non-holiday periods, while <xref ref-type="fig" rid="F8">Figure 8B</xref> displays the load forecast curve from May 1 to 3 May 2023, representing the forecast results for holiday periods. By comparing and analyzing <xref ref-type="fig" rid="F10">Figures 10A, B</xref>, it is evident that the power load in this area shows a noticeable upward trend during the holidays. Additionally, the load forecasting model proposed in this paper demonstrates superior performance in predicting power loads during both holiday and non-holiday periods.</p>
<p>To further validate the superior capabilities of the proposed forecasting models, supplementary experiments were conducted using a publicly available dataset from the Global Energy Forecasting Competition 2014 (GFC-2014), as referenced in research (<xref ref-type="bibr" rid="B12">Hong et al., 2016</xref>). The results of these additional forecasting experiments are presented in <xref ref-type="fig" rid="F11">Figure 11</xref>.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Comparison of forecasting results from different methods across different periods. <bold>(A)</bold> weekday load forecasting results. <bold>(B)</bold> holiday load forecasting results.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g011.tif"/>
</fig>
<p>And the statistic of the forecasting performance of these comparison methods are summarized as <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Forecast results derived from the GFC-2014 model.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Periods</th>
<th align="center">Models</th>
<th align="center">RMSE</th>
<th align="center">MAPE (%)</th>
<th align="center">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="center">Weekday</td>
<td align="center">LSTM</td>
<td align="center">71.3</td>
<td align="center">6.4</td>
<td align="center">0.81</td>
</tr>
<tr>
<td align="center">CNN-LSTM</td>
<td align="center">65.7</td>
<td align="center">5.5</td>
<td align="center">0.83</td>
</tr>
<tr>
<td align="center">ConvLSTM</td>
<td align="center">61.9</td>
<td align="center">5.2</td>
<td align="center">0.84</td>
</tr>
<tr>
<td align="center">ConvLSTM-3D</td>
<td align="center">68.8</td>
<td align="center">5.3</td>
<td align="center">0.88</td>
</tr>
<tr>
<td rowspan="4" align="center">Holiday</td>
<td align="center">LSTM</td>
<td align="center">61.7</td>
<td align="center">5.9</td>
<td align="center">0.87</td>
</tr>
<tr>
<td align="center">CNN-LSTM</td>
<td align="center">64.5</td>
<td align="center">6.8</td>
<td align="center">0.84</td>
</tr>
<tr>
<td align="center">ConvLSTM</td>
<td align="center">59.3</td>
<td align="center">3.9</td>
<td align="center">0.91</td>
</tr>
<tr>
<td align="center">ConvLSTM-3D</td>
<td align="center">58.5</td>
<td align="center">3.7</td>
<td align="center">0.92</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>It can be observed that the method ConvLSTM-3D proposed in this research get the best performance across different period. This is because the superior capability capturing the spatial-temporal correlations of the fusion feature datasets. By comparison, the LSTM get the worst performance across the holidays as the load consumption behaviors cannot be captured by solely temporal models. And the CNN-LSTM forecast better than LSTM, because the CNN module can fill this gap. ConvLSTM better captures spatiotemporal features by directly modeling spatial and temporal dependencies through convolutional operations, while CNN-LSTM may lose information by separating spatial and temporal modeling. And ConvLSTM-3D enhances ConvLSTM by using 3D convolutions to capture spatiotemporal dependencies simultaneously, providing superior modeling of complex dynamics in tasks like video processing.</p>
</sec>
<sec id="s4-3-3">
<title>4.3.3 Ablation study based on feature analysis</title>
<p>To further validate the performance improvement of the short-term load forecasting method based on time-weather fusion features proposed in this paper, a comparative analysis was conducted. This analysis includes time features based on Prophet feature components, weather features derived from weather components, and fusion features combining both time and weather components. The experiment was performed using identical model parameters and training data for consistency.</p>
<p>The load forecast results were compared for non-holiday periods (April 1st to April 3rd) and holiday periods (May 1st to May 3rd), with the forecast result curves presented in <xref ref-type="fig" rid="F12">Figure 12</xref>. Additionally, to further analyze the error distribution across different feature components, <xref ref-type="fig" rid="F13">Figure 13</xref> displays the average forecast error for the two forecast periods.</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption>
<p>Forecasting results for different feature components and different time period. <bold>(A)</bold> Prediction result curves under different feature quantities during the non-holiday period (April 1 to April 3). <bold>(B)</bold> Prediction result curves under different feature quantities during the holiday period (May 1 to March 3).</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g012.tif"/>
</fig>
<fig id="F13" position="float">
<label>FIGURE 13</label>
<caption>
<p>10 Average prediction error distribution under different feature components.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g013.tif"/>
</fig>
<p>The results indicate that the prediction error distribution when using only Prophet feature components or only weather feature components is relatively divergent, whereas the prediction error with time-weather fusion features is more concentrated. This observation can be attributed to the inherent differences in power load characteristics among users during different time periods in short-term load forecasting. For instance, residents&#x2019; power consumption habits vary significantly between weekdays and holidays and demonstrate a correlation with weather conditions. Employing a model that relies solely on weather features or time features may lead to underfitting, ultimately resulting in lower prediction accuracy. The specific prediction statistical indicators are summarized in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The experimental results of ablation based on feature analysis.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Forecast Region</th>
<th align="left">Period</th>
<th align="left">RMSE</th>
<th align="left">MAPE (%)</th>
<th align="left">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">Prophet Features</td>
<td align="left">April 1 - April 3</td>
<td align="left">70.2</td>
<td align="left">6.3</td>
<td align="left">0.82</td>
</tr>
<tr>
<td align="left">May 1 - May 3</td>
<td align="left">63.5</td>
<td align="left">5.8</td>
<td align="left">0.85</td>
</tr>
<tr>
<td rowspan="2" align="left">Weather Features</td>
<td align="left">April 1- April 3</td>
<td align="left">62.1</td>
<td align="left">5.4</td>
<td align="left">0.83</td>
</tr>
<tr>
<td align="left">May 1 - May 3</td>
<td align="left">67.4</td>
<td align="left">6.1</td>
<td align="left">0.75</td>
</tr>
<tr>
<td rowspan="2" align="left">Economic and Demographic Features</td>
<td align="left">April 1- April 3</td>
<td align="left">63.2</td>
<td align="left">5.8</td>
<td align="left">0.85</td>
</tr>
<tr>
<td align="left">May 1 - May 3</td>
<td align="left">65.2</td>
<td align="left">6.3</td>
<td align="left">0.82</td>
</tr>
<tr>
<td rowspan="2" align="left">Time-Weather Fusion Features</td>
<td align="left">April 1 - April 3</td>
<td align="left">58.2</td>
<td align="left">3.8</td>
<td align="left">0.92</td>
</tr>
<tr>
<td align="left">May 1 - May 3</td>
<td align="left">58.8</td>
<td align="left">3.6</td>
<td align="left">0.91</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>It is important to note that historical load data have been incorporated with the aforementioned features. The results demonstrated that the time-weather fusion features provided the most accurate forecasts across various periods, as this combination offers comprehensive insights into electrical consumption behaviors and the influence of weather conditions. In contrast, features relying solely on weather or social factors fail to capture a holistic representation of load consumption across different periods, neglecting variations in residents&#x2019; electricity usage patterns. The analysis of the experimental results clearly indicates that the proposed time-weather fusion feature significantly outperforms individual time-based or weather-based feature components.</p>
</sec>
<sec id="s4-3-4">
<title>4.3.4 Model robustness analysis</title>
<p>To verify the robustness and anti-interference capability of the model, 5-fold cross-validation was employed to evaluate the prediction results. Specifically, the dataset was divided into five distinct subsets. In each validation round, one of the subsets was designated as the validation set, while the remaining four subsets constituted the training set. The model was trained using the training set and subsequently evaluated using the validation set.</p>
<p>Performance was assessed using metrics such as mean square error and prediction accuracy, calculated after each validation round. This process yielded five sets of performance evaluation results, enabling analysis of the stability and generalization ability of the various prediction models. Since the robustness test assesses the performance of the model itself, independent of the training data, all comparison models utilized the same dataset. The validation was conducted using load data from regions 2 and 3, along with the corresponding meteorological data. The prediction results are presented in <xref ref-type="table" rid="T4">Table 4</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Five-fold cross validation test results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Forecast Region</th>
<th align="left">Fold</th>
<th align="left">RMSE</th>
<th align="left">MAPE (%)</th>
<th align="left">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="5" align="left">Region 2</td>
<td align="left">1</td>
<td align="left">58.3</td>
<td align="left">3.6</td>
<td align="left">0.93</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">60.2</td>
<td align="left">4.2</td>
<td align="left">0.91</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">59.1</td>
<td align="left">3.5</td>
<td align="left">0.92</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">63.5</td>
<td align="left">3.7</td>
<td align="left">0.92</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">64.1</td>
<td align="left">4.0</td>
<td align="left">0.93</td>
</tr>
<tr>
<td rowspan="5" align="left">Region 3</td>
<td align="left">1</td>
<td align="left">57.6</td>
<td align="left">4.0</td>
<td align="left">0.90</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">55.9</td>
<td align="left">4.1</td>
<td align="left">0.91</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">58.2</td>
<td align="left">3.9</td>
<td align="left">0.91</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">59.3</td>
<td align="left">3.9</td>
<td align="left">0.92</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">58.6</td>
<td align="left">4.2</td>
<td align="left">0.91</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The prediction accuracy of the short-term load forecasting method recommended in this paper matches the performance observed in the other two regions, demonstrating stability across different test and validation sets. This indicates that the method adopted in this paper is robust.</p>
<p>To thoroughly illustrate the improvements brought by the proposed method in short-term load forecasting, the experimental section compares and analyzes the prediction results of various models. Considering the differences between devices in different regions and the potential impacts of varying data sampling rates on new energy sources, this paper employs a downsampling method to enhance the model&#x2019;s prediction performance across different sampling rates. The results are presented in <xref ref-type="table" rid="T5">Table 5</xref>. Additionally, <italic>post hoc</italic> Nemenyi tests are used to statistically analyze the prediction performance across multiple cross-experiments, with the statistical results depicted in <xref ref-type="fig" rid="F14">Figure 14</xref>.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Performance comparison across different time intervals with various baseline models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Model category</th>
<th rowspan="2" align="left">Model name</th>
<th colspan="3" align="center">10&#xa0;min/point</th>
<th colspan="3" align="center">20&#xa0;min/point</th>
</tr>
<tr>
<th align="center">MAE (&#x2193;)</th>
<th align="center">RMSE (&#x2193;)</th>
<th align="center">MAPE (%) (&#x2193;)</th>
<th align="center">MAE (&#x2193;)</th>
<th align="center">RMSE (&#x2193;)</th>
<th align="center">MAPE (%) (&#x2193;)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="left">Statistical methods</td>
<td align="left">ARMA</td>
<td align="left">31.23</td>
<td align="left">48.23</td>
<td align="left">28.35</td>
<td align="left">35.32</td>
<td align="left">50.45</td>
<td align="left">31.36</td>
</tr>
<tr>
<td align="left">ARIMA</td>
<td align="left">30.62</td>
<td align="left">42.24</td>
<td align="left">25.22</td>
<td align="left">32.56</td>
<td align="left">46.36</td>
<td align="left">28.84</td>
</tr>
<tr>
<td align="left">HAR</td>
<td align="left">25.34</td>
<td align="left">40.15</td>
<td align="left">22.34</td>
<td align="left">26.32</td>
<td align="left">41.21</td>
<td align="left">24.43</td>
</tr>
<tr>
<td rowspan="3" align="left">Based on RNN model method</td>
<td align="left">LSTM</td>
<td align="left">14.38</td>
<td align="left">24.59</td>
<td align="left">12.13</td>
<td align="left">14.14</td>
<td align="left">23.84</td>
<td align="left">12.25</td>
</tr>
<tr>
<td align="left">GRU</td>
<td align="left">14.56</td>
<td align="left">25.13</td>
<td align="left">12.32</td>
<td align="left">14.52</td>
<td align="left">24.43</td>
<td align="left">12.33</td>
</tr>
<tr>
<td align="left">CNN-LSTM</td>
<td align="left">12.39</td>
<td align="left">22.64</td>
<td align="left">10.15</td>
<td align="left">14.41</td>
<td align="left">24.24</td>
<td align="left">11.14</td>
</tr>
<tr>
<td rowspan="4" align="left">Transformer model-based approach</td>
<td align="left">Transformer</td>
<td align="left">12.21</td>
<td align="left">20.05</td>
<td align="left">8.38</td>
<td align="left">12.24</td>
<td align="left">21.36</td>
<td align="left">9.96</td>
</tr>
<tr>
<td align="left">Informer</td>
<td align="left">12.35</td>
<td align="left">21.12</td>
<td align="left">8.35</td>
<td align="left">12.32</td>
<td align="left">20.12</td>
<td align="left">9.28</td>
</tr>
<tr>
<td align="left">Autoformer</td>
<td align="left">12.22</td>
<td align="left">20.47</td>
<td align="left">8.23</td>
<td align="left">12.14</td>
<td align="left">20.11</td>
<td align="left">9.15</td>
</tr>
<tr>
<td align="left">Methods in this article</td>
<td align="left">6.36</td>
<td align="left">14.13</td>
<td align="left">5.35</td>
<td align="left">7.62</td>
<td align="left">18.12</td>
<td align="left">8.14</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Model Category</th>
<th rowspan="2" align="left">Models</th>
<th colspan="3" align="center">30&#xa0;min/point</th>
<th colspan="3" align="center">60&#xa0;min/point</th>
</tr>
<tr>
<th align="center">MAE(&#x2193;)</th>
<th align="center">RMSE(&#x2193;)</th>
<th align="center">MAPE (%) (&#x2193;)</th>
<th align="center">MAE(&#x2193;)</th>
<th align="center">RMSE(&#x2193;)</th>
<th align="center">MAPE (%) (&#x2193;)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="left">Statistical methods</td>
<td align="left">ARMA</td>
<td align="left">40.24</td>
<td align="left">51.12</td>
<td align="left">38.45</td>
<td align="left">42.12</td>
<td align="left">55.75</td>
<td align="left">40.12</td>
</tr>
<tr>
<td align="left">ARIMA</td>
<td align="left">38.88</td>
<td align="left">50.35</td>
<td align="left">37.23</td>
<td align="left">40.35</td>
<td align="left">52.24</td>
<td align="left">38.88</td>
</tr>
<tr>
<td align="left">HAR</td>
<td align="left">35.25</td>
<td align="left">47.14</td>
<td align="left">35.12</td>
<td align="left">39.21</td>
<td align="left">50.21</td>
<td align="left">34.48</td>
</tr>
<tr>
<td rowspan="3" align="left">Based on RNN model method</td>
<td align="left">LSTM</td>
<td align="left">23.74</td>
<td align="left">40.35</td>
<td align="left">26.65</td>
<td align="left">30.46</td>
<td align="left">47.84</td>
<td align="left">28.84</td>
</tr>
<tr>
<td align="left">GRU</td>
<td align="left">24.12</td>
<td align="left">40.21</td>
<td align="left">26.56</td>
<td align="left">60.55</td>
<td align="left">92.37</td>
<td align="left">29.56</td>
</tr>
<tr>
<td align="left">CNN-LSTM</td>
<td align="left">21.11</td>
<td align="left">38.89</td>
<td align="left">22.54</td>
<td align="left">22.35</td>
<td align="left">35.56</td>
<td align="left">24.33</td>
</tr>
<tr>
<td rowspan="4" align="left">Transformer model-based approach</td>
<td align="left">Transformer</td>
<td align="left">20.78</td>
<td align="left">37.77</td>
<td align="left">21.36</td>
<td align="left">22.14</td>
<td align="left">30.25</td>
<td align="left">23.35</td>
</tr>
<tr>
<td align="left">Informer</td>
<td align="left">20.84</td>
<td align="left">37.55</td>
<td align="left">20.28</td>
<td align="left">21.12</td>
<td align="left">28.87</td>
<td align="left">23.01</td>
</tr>
<tr>
<td align="left">Autoformer</td>
<td align="left">19.89</td>
<td align="left">35.58</td>
<td align="left">19.92</td>
<td align="left">20.05</td>
<td align="left">27.90</td>
<td align="left">22.99</td>
</tr>
<tr>
<td align="left">Methods in this article</td>
<td align="left">15.44</td>
<td align="left">21.34</td>
<td align="left">14.22</td>
<td align="left">19.36</td>
<td align="left">35.21</td>
<td align="left">17.3</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F14" position="float">
<label>FIGURE 14</label>
<caption>
<p>Post-hoc Nemenyi statistics of prediction results of different prediction models.</p>
</caption>
<graphic xlink:href="fenrg-12-1501963-g014.tif"/>
</fig>
<p>The horizontal axis in the chart represents the average ranking of each method, displayed from right to left, with a color gradient transitioning from black to blue. If the average ranking difference reaches the critical difference (CD), it is highlighted with a red line. For instance, the proposed model in this paper significantly outperforms the GRU, RNN, HAR, ARIMA, and SARIMA models. Similarly, the Autoformer model shows significantly better performance than the RNN, HAR, ARIMA, and SARIMA models. These statistical results underscore the superiority and robustness of the method proposed in this paper for short-term load forecasting.</p>
</sec>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In existing short-term load forecasting approaches, the accuracy of load predictions across regions (both temporally and spatially) is often inadequate. This is primarily due to the lack of consideration for factors such as holiday load variations and differences in user electricity consumption behavior, making it challenging for prediction models to extract correlations among complex feature variables. This paper proposes a short-term load forecasting framework that integrates time and weather fusion features with a ConvLSTM-3D deep learning model.</p>
<p>The framework consists of two key components: the construction of time-weather fusion features and the development of the ConvLSTM-3D prediction model. In the first stage, the Prophet algorithm is employed to extract various time feature components, followed by the selection of important weather feature components using the SHAP algorithm. Finally, based on the importance of the different feature components, the selected time and weather features are reconstructed using an attention mechanism.</p>
<p>In the second stage, the traditional ConvLSTM model is enhanced to create a ConvLSTM-3D prediction model that is suitable for the fused features, allowing for effective training and prediction with the constructed fusion features. By comparing the load prediction results across different algorithms, the proposed method demonstrates advancements in short-term load forecasting performance and the robustness of the model.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>XY: Methodology, Software, Validation, Writing&#x2013;original draft. ShZ: Conceptualization, Formal Analysis, Investigation, Writing&#x2013;review and editing. KL: Project administration, Supervision, Validation, Writing&#x2013;review and editing. WC: Conceptualization, Data curation, Formal Analysis, Writing&#x2013;review and editing. SiZ: Project administration, Resources, Visualization, Writing&#x2013;review and editing. JC: Investigation, Validation, Visualization, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. The work is supported by State Grid Zhejiang Electric Power Co., LTD. Science and Technology project (B311SX23000C). The funder was not involved in the study design, collection, analysis, interpretation of data, the writing of this article, or the decision to submit it for publication.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>Authors XY, ShZ, KL, WC, SiZ, and JC were employed by State Grid Zhejiang Electric Power Co., Ltd Shaoxing Power Supply Company.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The authors declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdolrasol</surname>
<given-names>M. G. M.</given-names>
</name>
<name>
<surname>Hussain</surname>
<given-names>S. M. S.</given-names>
</name>
<name>
<surname>Ustun</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Sarker</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Hannan</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Mohamed</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Artificial neural networks based optimization techniques: a review</article-title>. <source>Electronics</source> <volume>10</volume> (<issue>21</issue>), <fpage>2689</fpage>. <pub-id pub-id-type="doi">10.3390/electronics10212689</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmad</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ghadi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Adnan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Load forecasting techniques for power system: research challenges and survey</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>71054</fpage>&#x2013;<lpage>71090</lpage>. <pub-id pub-id-type="doi">10.1109/access.2022.3187839</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alhussein</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Aurangzeb</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Haider</surname>
<given-names>S. I.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Hybrid CNN-LSTM model for short-term individual household load forecasting</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>180544</fpage>&#x2013;<lpage>180557</lpage>. <pub-id pub-id-type="doi">10.1109/access.2020.3028281</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bashir</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Haoyong</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tahir</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Liqiang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Short-term electricity load forecasting using hybrid prophet-LSTM model optimized by BPNN</article-title>. <source>Energy Rep.</source> <volume>8</volume>, <fpage>1678</fpage>&#x2013;<lpage>1686</lpage>. <pub-id pub-id-type="doi">10.1016/j.egyr.2021.12.067</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ceperic</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ceperic</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Baric</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>A strategy for short-term load forecasting by support vector regression machines</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>28</volume> (<issue>4</issue>), <fpage>4356</fpage>&#x2013;<lpage>4364</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2013.2269803</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2024a</year>). <article-title>Ultra-short term wind power prediction based on quadratic variational mode decomposition and multi-model fusion of deep learning</article-title>. <source>Comput. Electr. Eng.</source> <volume>116</volume>, <fpage>109157</fpage>. <pub-id pub-id-type="doi">10.1016/j.compeleceng.2024.109157</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C. L. P.</given-names>
</name>
</person-group> (<year>2024b</year>). <article-title>A survey on imbalanced learning: latest research, applications and future directions</article-title>. <source>Artif. Intell. Rev.</source> <volume>57</volume> (<issue>6</issue>), <fpage>137</fpage>&#x2013;<lpage>151</lpage>. <pub-id pub-id-type="doi">10.1007/s10462-024-10759-6</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dahl</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Brun</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kirsebom</surname>
<given-names>O. S.</given-names>
</name>
<name>
<surname>Andresen</surname>
<given-names>G. B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Improving short-term heat load forecasts with calendar and holiday data</article-title>. <source>Energies</source> <volume>11</volume> (<issue>7</issue>), <fpage>1678</fpage>. <pub-id pub-id-type="doi">10.3390/en11071678</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Benoit</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Foggia</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Besanger</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wurtz</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Neural network-based model design for short-term load forecast in distribution systems</article-title>. <source>IEEE Trans. power Syst.</source> <volume>31</volume> (<issue>1</issue>), <fpage>72</fpage>&#x2013;<lpage>81</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2015.2390132</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Short-term electricity-load forecasting by deep learning: a comprehensive survey</article-title>. <source>arXiv Prepr. arXiv:2408.16202</source>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Che</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Shahidehpour</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Machine-Learning based methods in short-term load forecasting</article-title>. <source>Electr. J.</source> <volume>34</volume> (<issue>1</issue>), <fpage>106884</fpage>. <pub-id pub-id-type="doi">10.1016/j.tej.2020.106884</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pinson</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zareipour</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Troccoli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hyndman</surname>
<given-names>R. J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Probabilistic energy forecasting: Global energy forecasting competition 2014 and beyond</article-title>. <source>Int. J. Forecast.</source> <volume>32</volume> (<issue>3</issue>), <fpage>896</fpage>&#x2013;<lpage>913</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijforecast.2016.02.001</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Gated spatial-temporal graph neural network based short-term load forecasting for wide-area multiple buses</article-title>. <source>Int. J. Electr. Power and Energy Syst.</source> <volume>145</volume>, <fpage>108651</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2022.108651</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A comparative study of the data-driven day-ahead hourly provincial load forecasting methods: from classical data mining to deep learning</article-title>. <source>Renew. Sustain. Energy Rev.</source> <volume>119</volume>, <fpage>109632</fpage>. <pub-id pub-id-type="doi">10.1016/j.rser.2019.109632</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Artificial combined model based on hybrid nonlinear neural network models and statistics linear models&#x2014;research and application for wind speed forecasting</article-title>. <source>Sustainability</source> <volume>10</volume> (<issue>12</issue>), <fpage>4601</fpage>. <pub-id pub-id-type="doi">10.3390/su10124601</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohammad</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>D. K.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>Y. C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Energy demand load forecasting for electric vehicle charging stations network based on convlstm and biconvlstm architectures</article-title>. <source>IEEE Access</source>, <volume>11</volume>: <fpage>67350</fpage>&#x2013;<lpage>67369</lpage>. <pub-id pub-id-type="doi">10.1109/access.2023.3274657</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moon</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jung</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rew</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rho</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hwang</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Combination of short-term load forecasting models based on a stacking ensemble approach</article-title>. <source>Energy Build.</source> <volume>216</volume>, <fpage>109921</fpage>. <pub-id pub-id-type="doi">10.1016/j.enbuild.2020.109921</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pijarski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Belowski</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Application of methods based on artificial intelligence and optimisation in power engineering&#x2014;introduction to the special issue</article-title>. <source>Energies</source> <volume>17</volume> (<issue>2</issue>), <fpage>516</fpage>. <pub-id pub-id-type="doi">10.3390/en17020516</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Si</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>China&#x2019;s urban energy system transition towards carbon neutrality: challenges and experience of Beijing and Suzhou</article-title>. <source>Renew. Sustain. Energy Rev.</source> <volume>183</volume>, <fpage>113468</fpage>. <pub-id pub-id-type="doi">10.1016/j.rser.2023.113468</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Mi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Short-term solar irradiance forecasting model based on artificial neural network using statistical feature parameters</article-title>. <source>Energies</source> <volume>5</volume> (<issue>5</issue>), <fpage>1355</fpage>&#x2013;<lpage>1370</lpage>. <pub-id pub-id-type="doi">10.3390/en5051355</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Short-term load forecasting for industrial customers based on TCN-LightGBM</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>36</volume> (<issue>3</issue>), <fpage>1984</fpage>&#x2013;<lpage>1997</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2020.3028133</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Short-term load forecasting of industrial customers based on SVMD and XGBoost</article-title>. <source>Int. J. Electr. Power and Energy Syst.</source> <volume>129</volume>, <fpage>106830</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2021.106830</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wazirali</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yaghoubi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Abujazar</surname>
<given-names>M. S. S.</given-names>
</name>
<name>
<surname>Ahmad</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Vakili</surname>
<given-names>A. H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>State-of-the-art review on energy and load forecasting in microgrids using artificial neural networks, machine learning, and deep learning techniques</article-title>. <source>Electr. power Syst. Res.</source> <volume>225</volume>, <fpage>109792</fpage>. <pub-id pub-id-type="doi">10.1016/j.epsr.2023.109792</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Digital transformation and rule of law based on peak CO2 emissions and carbon neutrality</article-title>. <source>Sustainability</source> <volume>14</volume> (<issue>12</issue>), <fpage>7487</fpage>. <pub-id pub-id-type="doi">10.3390/su14127487</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Bi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Multi-view broad learning system for electricity theft detection</article-title>. <source>Appl. Energy</source> <volume>352</volume>, <fpage>121914</fpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2023.121914</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C. L. P.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Solving the imbalanced problem by metric learning and oversampling</article-title>. <source>IEEE Trans. Knowl. Data Eng.</source> <volume>36</volume>, <fpage>9294</fpage>&#x2013;<lpage>9307</lpage>. <pub-id pub-id-type="doi">10.1109/tkde.2024.3419834</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Kou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Reliable long-term energy load trend prediction model for smart grid using hierarchical decomposition self-attention network</article-title>. <source>IEEE Trans. Reliab.</source> <volume>72</volume> (<issue>2</issue>), <fpage>609</fpage>&#x2013;<lpage>621</lpage>. <pub-id pub-id-type="doi">10.1109/tr.2022.3174093</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhuang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A multi-scale spatial-temporal graph neural network-based method of multienergy load forecasting in integrated energy system</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>15</volume>, <fpage>2652</fpage>&#x2013;<lpage>2666</lpage>. <pub-id pub-id-type="doi">10.1109/tsg.2023.3315750</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>