<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Energy Res.</journal-id>
<journal-title>Frontiers in Energy Research</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Energy Res.</abbrev-journal-title>
<issn pub-type="epub">2296-598X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">773805</article-id>
<article-id pub-id-type="doi">10.3389/fenrg.2021.773805</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Energy Research</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Electricity Theft Detection in Power Consumption Data Based on Adaptive Tuning Recurrent Neural Network</article-title>
<alt-title alt-title-type="left-running-head">Lin et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">Electricity Theft Detection by TSRNN Model</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Lin</surname>
<given-names>Guoying</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Feng</surname>
<given-names>Haoyang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Feng</surname>
<given-names>Xiaofeng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wen</surname>
<given-names>Hongwu</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Yuanzheng</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Hong</surname>
<given-names>Shaoyong</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1471140/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ni</surname>
<given-names>Zhixian</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1475122/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<label>
<sup>1</sup>
</label>Metrology Center of Guangdong Power Grid Corporation, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<label>
<sup>2</sup>
</label>College of Electrical Engineering, Zhejiang University, <addr-line>Hangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<label>
<sup>3</sup>
</label>Zhanjiang Power Supply Bureau of Guangdong Power Grid Co. Ltd., <addr-line>Zhanjiang</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<label>
<sup>4</sup>
</label>China-EU Institute for Clean and Renewable Energy, Huazhong University of Science and Technology, <addr-line>Wuhan</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<label>
<sup>5</sup>
</label>School of Data Science, Guangzhou Huashang College, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1256586/overview">Bin Zhou</ext-link>, Hunan University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1478084/overview">Huazhou Chen</ext-link>, Guilin University of Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1429899/overview">Xueqian Fu</ext-link>, China Agricultural University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1480600/overview">Chao Yuan</ext-link>, Hunan University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1057890/overview">Jun Zeng</ext-link>, South China University of Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Shaoyong Hong, <email>shy2002021@163.com</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Process and Energy Systems Engineering, a section of the journal Frontiers in Energy Research</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>10</day>
<month>11</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>9</volume>
<elocation-id>773805</elocation-id>
<history>
<date date-type="received">
<day>10</day>
<month>09</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>10</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2021 Lin, Feng, Feng, Wen, Li, Hong and Ni.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Lin, Feng, Feng, Wen, Li, Hong and Ni</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>Electricity theft behavior has serious influence on the normal operation of power grid and the economic benefits of power enterprises. Intelligent anti-power-theft algorithm is required for monitoring the power consumption data to recognize electricity power theft. In this paper, an adaptive time-series recurrent neural network (TSRNN) architecture was built up to detect the abnormal users (i.e.,&#x20;the electricity theft users) in time-series data of the power consumption. In fusion with the synthetic minority oversampling technique (SMOTE) algorithm, a batch of virtual abnormal observations were generated as the implementation for training the TSRNN model. The power consumption record was characterized with the sharp data (ARP), the peak data (PEA), and the shoulder data (SHO). In the TSRNN architectural framework, a basic network unit was formed with three input nodes linked to one hidden neuron for extracting data features from the three characteristic variables. For time-series analysis, the TSRNN structure was re-formed by circulating the basic unit. Each hidden node was designed receiving data from both the current input neurons and the time-former neuron, thus to form a combination of network linking weights for adaptive tuning. The optimization of the TSRNN model is to automatically search for the most suitable values of these linking weights driven by the collected and simulated data. The TSRNN model was trained and optimized with a high discriminant accuracy of 95.1%, and evaluated to have 89.3% accuracy. Finally, the optimized TSRNN model was used to predict the 47 real abnormal samples, resulting in having only three samples false predicted. These experimental results indicated that the proposed adaptive TSRNN architecture combined with SMOTE is feasible to identify the abnormal electricity theft behavior. It is prospective to be applied to online monitoring of distributed analysis of large-scale electricity power consumption&#x20;data.</p>
</abstract>
<kwd-group>
<kwd>electricity theft</kwd>
<kwd>TSRNN</kwd>
<kwd>adaptive parameter tuning</kwd>
<kwd>intelligent learning</kwd>
<kwd>SMOTE</kwd>
<kwd>power consumption data</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>With the increasing scale of the power grid, the power consumption is becoming larger year by year. People are concerning on the economic operation of power network, saving of electric resources, reduction of grid line loss, and structural optimization on power consumption (<xref ref-type="bibr" rid="B11">Dileep, 2020</xref>). However, the customer&#x2019;s behavior of stealing electricity comes in non-stopping emergence. This infraction phenomenon has seriously affected the normal operation of power grid and the economic benefits of power enterprises (<xref ref-type="bibr" rid="B19">Li et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B31">Zhang et&#x20;al., 2020</xref>). The electricity theft rate in developing countries is as high as 30%, and the social power supply and consumption has also been greatly influenced. According to rough statistics, China&#x2019;s power enterprises lose as much as 20 billion CNY every year due to power theft. Therefore, power enterprises must carry out efficient anti-electricity-theft work, in order to guarantee the reasonable power supply and rational use of electricity, thus to reduce economic losses as much as possible (<xref ref-type="bibr" rid="B4">Aryanezhad, 2019</xref>).</p>
<p>The traditional detection methods of power theft mainly rely on the scheduled operations of technicians who work in power supply enterprises. The operation goes with reading the electricity meter and then recording, counting, and performing manual analysis and calculation. In the hardware aspect, there are multifaceted operations that can prevent energy theft, such as to install the specialized watt-hour metering box, to implement a kind of conductor that closes the low-voltage outlet to the metering device, to add anti-thief function to the watt-hour meter, and to improve the application rate of electrical acquisition system (<xref ref-type="bibr" rid="B18">Jokar et&#x20;al., 2016</xref>). However, most of these traditional anti-theft detection methods focus on the improvement of power devices. There is a lack of sufficient anti-power-stealing algorithms to analyze massive historical power consumption data, so it is difficult to find the power consumption characteristics of power-stealing users and detect the power-stealing behavior realized by advanced attack means (<xref ref-type="bibr" rid="B1">Ahmad et&#x20;al., 2015</xref>). Therefore, the development of power industry needs to strengthen the development of new artificial intelligence and information and automation technology. With the continuous improvement of dynamic monitoring and acquisition technology of power consumption data of power grid users, it is of great engineering significance to study the intelligent anti-power-theft algorithm based on the big data of the power consumption to identify the power theft behavior (<xref ref-type="bibr" rid="B26">Ren et&#x20;al., 2020</xref>; <xref ref-type="bibr" rid="B30">Zhang et&#x20;al., 2021</xref>).</p>
<p>At present, the most popular scheme is to lay out the smart grid detection architecture and framework, then to collect the power consumption data, and upload them to the centralized data processing center through the terminal smart meter, and successively, the centralized data can be further analyzed by intelligent algorithms to detect electricity theft. The prevalent anti-power-stealing data mining algorithms include clustering, BP neural network, and local outlier detection algorithm (<xref ref-type="bibr" rid="B2">Al-Dahidi et&#x20;al., 2019</xref>; <xref ref-type="bibr" rid="B21">Li Y. et&#x20;al., 2021</xref>). Many practical experiments have been studied in previous research works. A typical load curve is extracted from the power consumption data by applying the adaptive K-means clustering algorithm to realize load forecasting and load control (<xref ref-type="bibr" rid="B33">Zhu et&#x20;al., 2016</xref>). The situation of abnormal point detection method was proposed based on a fuzzy neural network to deal with various data, which provides a new idea for mining abnormal data from the power consumption records (<xref ref-type="bibr" rid="B25">Mozaffar et&#x20;al., 2018</xref>). The flying anomaly factor detection and analysis method was investigated to detect an electric energy meter flying anomaly (<xref ref-type="bibr" rid="B22">Li et&#x20;al., 2016</xref>). A novel detection method of power theft was constructed based on the one-class SVM algorithm. A calibration model was established by analyzing a large number of historical data. If the current data are inconsistent with the model, it is considered that there is a possibility of power theft (<xref ref-type="bibr" rid="B12">Dou et&#x20;al., 2018</xref>). Also, the RBF neural network was proposed to detect the electricity-stealing behavior, which used the data characteristics of voltage, current, and power factor to detect electricity theft, to make a positive detection on electricity stealing (<xref ref-type="bibr" rid="B6">Cao et&#x20;al., 2018</xref>).</p>
<p>Due to the wide layout of the power grid, the large-scale deployment of smart meters should consume a lot of resources. In order to save the energy consumption of distributed terminal nodes, and reduce the non-essential data transmission, it is necessary to study modern data mining technology, in integration with machine learning algorithms (<xref ref-type="bibr" rid="B28">Wang et&#x20;al., 2020</xref>; <xref ref-type="bibr" rid="B23">Li Z. et&#x20;al., 2021</xref>). The application of indirect data anomaly detection as well as some preprocessing and analyzing technologies is much necessary to achieve the online detection of power theft. However, data-driven power theft detection is a special type of anomaly detection, which has a serious class imbalance problem (<xref ref-type="bibr" rid="B5">Avila et&#x20;al., 2018</xref>). Actually, the number of normal power consumption users is much larger than the number of abnormal users. The inherent imbalance of data will affect the performance of traditional machine learning methods. Until now, only a few studies have considered the category imbalance in power theft detection (<xref ref-type="bibr" rid="B29">Zhang et&#x20;al., 2019</xref>). The solutions of these works are mainly performed with undersampling and oversampling methods in the aspects of data analytical algorithm. They were keen on simultaneously implementing the random oversampling and undersampling techniques, to select the best detection effect by testing different sampling ratios. Otherwise, they focus on increasing the misclassification cost of abnormal users to improve the detection rate of electricity theft, by setting penalty parameters for support vector machine misclassification of normal and abnormal users (<xref ref-type="bibr" rid="B16">Hu et&#x20;al., 2019</xref>).</p>
<p>Generally, the electricity theft monitoring data are a kind of time-series data. The difficulty of data analysis lies in how to find the abnormal data from the constantly updated dynamic data flow, so as to accurately predict the theft users. The fact that the data are extremely imbalance is the first-of-all analytical difficulty. Many experiments have proved that oversampling is a solution to the category imbalance problem. In essence, the random oversampling method increases the weight in the sample set by randomly copying a few samples. It does not increase classification accuracy but is easy to cause over-fitting (<xref ref-type="bibr" rid="B15">He and Garcia, 2019</xref>). Synthetic minority oversampling technique (SMOTE) is an unbalanced data recall method that is improved from the linear interpolation calculation methodology. It uses the local prior distribution information of samples to improve the accuracy of minority samples, to solve the data imbalance problem (<xref ref-type="bibr" rid="B32">Zhu et&#x20;al., 2017</xref>). Furthermore, the recurrent neural network (RNN) is an effective intelligent machine learning method that is especially effective for monitoring and analyzing time-series dynamic data flow. The RNN is derived from the conventional fully connected neural network (FCNN) model. Its core operation is to compute the result of each neuron not only from its input data (similar to the FCNN) but also from the historical variables from its former calculations (different from the FCNN). The RNN model is widely used in addressing the tasks of sequential data processing (<xref ref-type="bibr" rid="B24">Liu et&#x20;al., 2020</xref>). The running of the RNN structure is to produce a neuron output by combined fusing of the current status data with the previous status data of the system. The RNN is able to automatically learn the time correlation of the input data without specifying any lag observations (<xref ref-type="bibr" rid="B10">Cossu et&#x20;al., 2021</xref>). It is well known that the traditional time-series analytical methods (such as auto-correlation) need to identify the seasonality and stability from the time-series data. The effectiveness of identification may vary according to the network structure and the calculation speed, and it needs to be adjusted for each simulation (<xref ref-type="bibr" rid="B8">Chen et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B13">Farjaminezhad et&#x20;al., 2021</xref>). The characteristic of the RNN is to create a closed-loop calculation in the hidden layer, which forms a circulating adaptive model to capture the internal hidden historical state features in the way of iterative update, and thus to complete the process of error level accumulation in the training stage. In effect, the RNN model is enforced to adapt the error accumulation and improve the model robustness (<xref ref-type="bibr" rid="B27">St&#xe5;hl et&#x20;al., 2019</xref>).</p>
<p>This paper is aimed at designing a data-driven adaptive parameter optimization time-series RNN (TSRNN) architecture, for intelligent machine learning to solve the problem of abnormal monitoring of power consumption. The TSRNN architecture with an adaptive training strategy is constructed by monitoring, collecting, and analyzing the observed data of a stage. Then, the non-linear features of the observed data can be extracted by developing a hyperparametric optimization mode of RNN, in fusion with a SMOTE solvation of data imbalance. On this algorithmic basis, the power-stealing users with abnormal characteristics are identified in a large number of power user samples. In structural detail, grid search is designed for the parameter selection of the RNN linking weights, and also, a fault-tolerance iteration mechanism is adopted for parameter optimization in the closed-loop training stage, to control the error accumulation in model prediction, so as to enhance the model robustness. In this way, the proposed intelligent TSRNN architecture with data-driven adaptive parameter optimization is validated through data training and prediction. The optimized model is effective for accurate extraction of the data features of power-stealing behavior. The establishment of the intelligent TSRNN model is expected to overcome the costly, laborious, and time-consuming problems of the traditional methods for monitoring electricity theft. It is feasible to speed up to locate the abnormal watt-hour meter terminals and accurately identify the power-stealing users. The proposed method helps promote the development of artificial intelligence and information analysis technology in the field of power grid operation and maintenance.</p>
</sec>
<sec id="s2">
<title>Methodologies</title>
<p>In this section, we discuss the basic structure of the TSRNN architecture and the algorithmic progress of SMOTE balancing. The energy theft detection model is established and further optimized by fusion of TSRNN and SMOTE. And the discriminant indicators are introduced based on the confusion matrix for the quasi-qualitative recognition of the abnormal user&#x20;data.</p>
<sec id="s2-1">
<title>The Principle of SMOTE</title>
<p>The SMOTE algorithm is an oversampling method based on synthetic sampling proposed by Chawla (<xref ref-type="bibr" rid="B7">Chawla et&#x20;al., 2002</xref>). In geometric sense, the SMOTE method firstly observes the minority samples and connects them and a batch of their surrounding samples. Then, it produces new samples by random insertion on the connecting lines. The connection and insertion operation can reduce the imbalance of sample space and simultaneously prevent the over-fitting phenomenon by suppressing too large repetition of the original minority samples (<xref ref-type="bibr" rid="B14">Fern&#xe1;ndez et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B9">Chen et&#x20;al., 2021</xref>). The schematic diagram for generating new samples by the SMOTE algorithm is shown in <xref ref-type="fig" rid="F1">Figure&#x20;1</xref>. Specifically, the SMOTE sample-generating procedures are described in the following steps:<list list-type="simple">
<list-item>
<p>Step 1: Let <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> be the minority samples and set the sampling number <inline-formula id="inf2">
<mml:math id="m2">
<mml:mi>r</mml:mi>
</mml:math>
</inline-formula> according to the number ratio of the majority samples over the minority samples</p>
</list-item>
<list-item>
<p>Step 2: Search <inline-formula id="inf3">
<mml:math id="m3">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula> samples in the neighborhood of the minority samples, where <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>Step 3: Randomly select <inline-formula id="inf5">
<mml:math id="m5">
<mml:mi>r</mml:mi>
</mml:math>
</inline-formula> samples from the <inline-formula id="inf6">
<mml:math id="m6">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula> neighborhood sample, to form the neighborhood sample set <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>Step 4: To generate a set of new samples by random linear interpolation computation, the new samples are denoted as <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>&#x2026;</mml:mo>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where</p>
</list-item>
</list>
<disp-formula id="e1">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>rand</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2217;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>&#x2026;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>with <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mtext>rand</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> representing a random number in the interval of <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mn>0,1</mml:mn>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. Then, <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is regarded as the algorithmic implementation of the minority samples.<list list-type="simple">
<list-item>
<p>Step 5: The newly generated samples <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are regarded as the algorithmic implementation of the minority samples, added to the original sample set to form a brand new training sample set together with the majority samples.</p>
</list-item>
</list>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Schematic diagram of the SMOTE algorithm.</p>
</caption>
<graphic xlink:href="fenrg-09-773805-g001.tif"/>
</fig>
<p>The SMOTE algorithm makes artificial synthesis of minority samples by random interpolation. Compared with the traditional methods of random replication, SMOTE reduces redundant information of newly generated minority samples and effectively avoids the phenomenon of over-fitting in the subsequent data mining processes. In algorithm, SMOTE shows its uncertainty in part of selecting the nearest neighborhood of the original minority samples, namely, the number of neighbor samples (i.e.,&#x20;the number of <inline-formula id="inf13">
<mml:math id="m14">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>) has a great influence on the model performance. When SMOTE is embedded in fusion with the TSRNN architecture, the number of neighbor samples would be designed as one of the tunable parameters for the network model optimization.</p>
</sec>
<sec id="s2-2">
<title>Time-Series RNN Model</title>
<p>The data-driven time-series analysis problem is theoretically described as a general ordinary differential model (<xref ref-type="bibr" rid="B20">Li and Yang, 2021</xref>), formulated as<disp-formula id="e2">
<mml:math id="m15">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">z</mml:mi>
<mml:mo mathvariant="bold">&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo mathvariant="bold">&#x3d;</mml:mo>
<mml:mi mathvariant="bold">f</mml:mi>
<mml:mrow>
<mml:mo mathvariant="bold">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mo mathvariant="bold">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:mi mathvariant="bold">z</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mtext>d</mml:mtext>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the current state of the system and <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mtext>R</mml:mtext>
<mml:mtext>d</mml:mtext>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the instant input data. In common sense, the model function <inline-formula id="inf16">
<mml:math id="m18">
<mml:mi>f</mml:mi>
</mml:math>
</inline-formula> is unknown, but it can be estimated by simulation on the discrete observation of the current state <inline-formula id="inf17">
<mml:math id="m19">
<mml:mi mathvariant="bold">z</mml:mi>
</mml:math>
</inline-formula> and the instant input <inline-formula id="inf18">
<mml:math id="m20">
<mml:mi mathvariant="bold">x</mml:mi>
</mml:math>
</inline-formula>. On these lines, the fully connected neural network (FCNN) is suitable to resolve the data-driven analytical models.</p>
<p>An FCNN module is traditionally applied as a black box to directly transform the input data to the hidden layer and then to get the output. The generated data acquired at each neuron node are described as <inline-formula id="inf19">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where the activation function <inline-formula id="inf20">
<mml:math id="m22">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is usually a kind of simple linear transformation, while the operation inside the FCNN has no physical interpretations. The black-box model may not be able to capture the detailed data transition in the time series. The TSRNN is proposed to solve this&#x20;issue.</p>
<p>The TSRNN architecture is built up with circulation computation of the hidden layer. To unfold the circulation ring, the TSRNN structure is introduced as shown in <xref ref-type="fig" rid="F2">Figure&#x20;2</xref>. As is shown in <xref ref-type="fig" rid="F2">Figure&#x20;2</xref>, the TSRNN architecture is supposed to be constructed along a time variance axis. At the starting of time, the power consumption user data are input into the network and delivered to the first hidden layer (<italic>H</italic>
<sub>1</sub>) while <inline-formula id="inf21">
<mml:math id="m23">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The data are transformed and calculated to extract the first level of neural features and then delivered to the next hidden layer when <inline-formula id="inf22">
<mml:math id="m24">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> varies. At each time step, the result of each neuron computation depends not only on the current input but also on the computation results. In this way, the TSRNN captures the intercorrelation between the time longitudinal parameters and the section parameters. As such, there are two network linking weight effects: one describes the direct effect from network layer delivery and the other shows the indirect data influence from the time-series circulation of the hidden layers. Any change in the direct weights or in the indirect weights will cause a change in the output at any instant moment of time (<xref ref-type="bibr" rid="B3">Alkinani et&#x20;al., 2021</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Structural design of the time-series RNN (TSRNN) architecture.</p>
</caption>
<graphic xlink:href="fenrg-09-773805-g002.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F2">Figure&#x20;2</xref> also presents a simple TSRNN cell structure at the instant moment of time <inline-formula id="inf23">
<mml:math id="m25">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. To be specific, a TSRNN cell is actually a single layer of hidden neurons. This hidden layer is denoted as <inline-formula id="inf24">
<mml:math id="m26">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and there are many hidden neurons for functional calculation, i.e.,&#x20;<inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>&#x2026;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Suppose the current input data are <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>&#x2026;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> from the power consumption user data, regarded as the direct input. The time-lag input data are acquired from the network calculation in the hidden layer <inline-formula id="inf27">
<mml:math id="m29">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> at the time moment of <inline-formula id="inf28">
<mml:math id="m30">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, taken as the indirect input. Then, <inline-formula id="inf29">
<mml:math id="m31">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> works as a <inline-formula id="inf30">
<mml:math id="m32">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula>-time hidden layer to extract data features from the direct inputs as well as the indirect inputs. The output of <inline-formula id="inf31">
<mml:math id="m33">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is influenced by both <inline-formula id="inf32">
<mml:math id="m34">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf33">
<mml:math id="m35">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. It can be formulated as<disp-formula id="e3">
<mml:math id="m36">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">W</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold-italic">U</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where the function <inline-formula id="inf34">
<mml:math id="m37">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> simply represents the sigmoid function which would strictly limit the transformed features in the standard variable range of <inline-formula id="inf35">
<mml:math id="m38">
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1,1</mml:mn>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. The parameters <inline-formula id="inf36">
<mml:math id="m39">
<mml:mi mathvariant="bold-italic">W</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf37">
<mml:math id="m40">
<mml:mi mathvariant="bold-italic">U</mml:mi>
</mml:math>
</inline-formula> represent the linking weights for data connection and for the time variance connection, respectively.</p>
<p>Successively, data <inline-formula id="inf38">
<mml:math id="m41">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, namely, the set of feature data included in <inline-formula id="inf39">
<mml:math id="m42">
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, are further delivered to a softmax unit for discriminant calculation. Thus, the neural network output at&#x20;the&#x20;time-series moment of <inline-formula id="inf40">
<mml:math id="m43">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> is mathematically demonstrated as<disp-formula id="e4">
<mml:math id="m44">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">V</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf41">
<mml:math id="m45">
<mml:mi mathvariant="bold-italic">V</mml:mi>
</mml:math>
</inline-formula> represents the linking weights involving the data transform from <inline-formula id="inf42">
<mml:math id="m46">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf43">
<mml:math id="m47">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and the function <inline-formula id="inf44">
<mml:math id="m48">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>&#x22c5;</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> operates the <italic>k</italic>-means clustering by Mahalanobis distance<disp-formula id="e5">
<mml:math id="m49">
<mml:mrow>
<mml:mtext>mah</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mi mathvariant="bold-italic">j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mtext>T</mml:mtext>
</mml:msup>
<mml:msup>
<mml:mi>&#x3a3;</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
<mml:mtext>&#xa0;&#xa0;for&#xa0;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>&#x2026;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>The Mahalanobis distance between any two of the <inline-formula id="inf45">
<mml:math id="m50">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> samples is calculated according to <xref ref-type="disp-formula" rid="e5">Eq. 5</xref> and then to obtain the distance matrix <inline-formula id="inf46">
<mml:math id="m51">
<mml:mrow>
<mml:mi mathvariant="bold">KM</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> at the instant time moment of <inline-formula id="inf47">
<mml:math id="m52">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula>, namely,<disp-formula id="e6">
<mml:math id="m53">
<mml:mrow>
<mml:mi mathvariant="bold">KM</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>mah</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>11</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>mah</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>12</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>mah</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>21</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>mah</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>22</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>mah</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>mah</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>mah</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>mah</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22f1;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>mah</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mi>&#x22c5;</mml:mi>
<mml:mtext>&#xa0;for&#xa0;</mml:mtext>
<mml:mi>t</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mtext>start</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mtext>end</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf48">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>mah</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x225c;</mml:mo>
<mml:mtext>mah</mml:mtext>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">O</mml:mi>
<mml:mi mathvariant="bold-italic">j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Finally, the Mahalanobis-based k-means clustering results of the TSRNN-extracted feature data are used for further calculation of the discriminant indicators, thus to help identify the abnormal users from all of the electric power consumption&#x20;data.</p>
</sec>
<sec id="s2-3">
<title>Discriminant Indicators</title>
<p>The power consumption data are originally imbalanced because the normal electricity users are much larger than the electricity thieves. It is expensive to identify the abnormal users. In our algorithmic designs, SMOTE is functional to alleviate the data imbalance, and the adaptive TSRNN model extracts the feature of power consumption data for improving the model discrimination accuracy with the k-means Mahalanobis measure. The model should be evaluated with quantitative indicators. The confusion matrix is a basic tool to evaluate the model performance (see <xref ref-type="table" rid="T1">Table&#x20;1</xref>). Then, the indicators of each model are verified based on the matrix&#x20;table.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Confusion matrix for evaluation of the discrimination/classification models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" colspan="2" align="center"/>
<th colspan="2" align="center">Prediction</th>
</tr>
<tr>
<th align="center">Positive</th>
<th align="center">Negative</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">Actual data</td>
<td align="left">Positive</td>
<td align="left">True positive (TP)</td>
<td align="left">False negative (FN)</td>
</tr>
<tr>
<td align="left">Negative</td>
<td align="left">False positive (FP)</td>
<td align="left">True negative (TN)</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>By definition of the confusion matrix, the normal power consumption users are distinguished as the negative records, while the abnormal users are taken as positive. Thus, the table markers are interpreted with the following information: <list list-type="simple">
<list-item>
<p>- TP indicates that the abnormal user (positive) is accurately predicted as abnormal (positive),</p>
</list-item>
<list-item>
<p>- TN indicates that the normal user (negative) is accurately predicted as normal (negative),</p>
</list-item>
<list-item>
<p>- FP indicates that the actual normal user (negative) is predicted false as abnormal (positive),</p>
</list-item>
<list-item>
<p>- FN indicates that the actual abnormal user (positive) is predicted false as normal (negative).</p>
</list-item>
</list>
</p>
<p>Multiple indicators are further calculated according to the confusion matrix, such as the classification accuracy (<inline-formula id="inf49">
<mml:math id="m55">
<mml:mrow>
<mml:mtext>ACC</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>), true positive rate (<inline-formula id="inf50">
<mml:math id="m56">
<mml:mrow>
<mml:mtext>TPR</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>), and false alarm rate (<inline-formula id="inf51">
<mml:math id="m57">
<mml:mrow>
<mml:mtext>FAR</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>). The calculations are presented as follows:<disp-formula id="e7">
<mml:math id="m58">
<mml:mrow>
<mml:mtext>ACC</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>TN</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>TN</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m59">
<mml:mrow>
<mml:mtext>TPR</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="e9">
<mml:math id="m60">
<mml:mrow>
<mml:mtext>FAR</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>FP</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>FP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>TN</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
<p>These indicators are used to evaluate the model performance of the adaptive parametric-scaling TSRNN architecture. It is learnt from <xref ref-type="disp-formula" rid="e7">Eqs. 7</xref>&#x2013;<xref ref-type="disp-formula" rid="e9">9</xref> that the higher the TP and TN are, the better the model performance&#x20;is.</p>
<p>For fault-tolerant analysis, the model prediction results can be monitored at every moment of the dynamic changing time series. By data export, there are a series of prediction results acquired for the model classification of normal and abnormal users. Then, the frequency of identification of abnormal is counted for each user over the whole time-series axis, thus&#x20;to&#x20;provide an extra confirmation of the model predictions.</p>
</sec>
</sec>
<sec id="s3">
<title>Analysis of Power Consumption Data</title>
<p>A total of 929 electricity/power consumption users were monitored continuously from January 1, 2017, to March 31, 2019, with the minimum time changing unit of 1&#xa0;day; thus, we recorded 820 instant moments in the long time series spanning 25&#xa0;months. Their electricity use data were collected in different partitions of time periods of hours according to the total usage amount. In detail, the electricity used during the hours of 00:00&#x2013;08:00 is named the off-peak data (denoted as OPE for short), during 08:00&#x2013;12:00 as the peak data (PEA), during 18:00&#x2013;22:00 as the sharp data (ARP), and during the rest hours as the shoulder data (SHO).</p>
<p>If the electricity users are taken as the analytical samples, the power consumption characteristics of the 929 samples are demonstrated by the recorded data of OPE, PEA ARP, and SHO. There are 820 digital records for each user by time variance. As the maximum record is over thirty thousand and the minimum record is zero, the dataset should be normalized before analysis, applying the min&#x2013;max normalization method (<xref ref-type="bibr" rid="B17">Jin et&#x20;al., 2015</xref>). Then, we statistically derived the sample distribution using the average electricity consumption of the 820&#x20;time nodes (see <xref ref-type="fig" rid="F3">Figure&#x20;3</xref>). As is seen from <xref ref-type="fig" rid="F3">Figure&#x20;3</xref>, the users do not use electricity all along time; for example, some electricity consumption appears high in the ARP time but low or even zero in SHO, and some goes high in PEA but zero in ARP or OPE. To be specific, it is seen from the sub-figure of OPE (the blue plot) that only one user out of the 929 keeps using electricity during the OPE time period. Thus, it is recognized with statistical principles that the OPE data property hardly provides data information for discriminating the abnormal users. Then, the OPE data do not participate in the following modeling processes of SMOTE balancing and TSRNN training.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Statistical descriptive plots of the power consumption data in different time period partitions.</p>
</caption>
<graphic xlink:href="fenrg-09-773805-g003.tif"/>
</fig>
</sec>
<sec id="s4">
<title>Data Balancing by SMOTE</title>
<p>Practically, we have the priori target classification index for the 929 available power consumption user samples. There are originally 882 normal samples and only 47 abnormal samples. The normal samples are the majority, and the abnormal ones are the minority. The imbalance ratio of the normal over the abnormal goes to a great extent of around 19:1. The scattering distribution of the 929 samples is a plot in the 3D axis based on the three basic variables of ARP, PEA, and SHO (see <xref ref-type="fig" rid="F4">Figure&#x20;4A</xref>). To ease the heavy imbalance status, the SMOTE algorithm is applied to increase the proportion of the minority samples by linear interpolations. According to the principle of the SMOTE simulation as introduced in <italic>The Principle of SMOTE</italic>, a batch of virtual samples are generated by interpolations on the original 47 abnormal samples.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Distribution of the power consumption user samples (panel <bold>(A)</bold> is for the original 929 samples, and panel <bold>(B)</bold> is for the SMOTE-balanced output of the 1,151 samples).</p>
</caption>
<graphic xlink:href="fenrg-09-773805-g004.tif"/>
</fig>
<p>Theoretically, one virtual sample is generated from the linking edge of every two samples. The 47 available samples are able to generate 1,081 (i.e.,&#x20;<inline-formula id="inf52">
<mml:math id="m61">
<mml:mrow>
<mml:msubsup>
<mml:mtext>C</mml:mtext>
<mml:mrow>
<mml:mn>47</mml:mn>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) new samples in all, from which we randomly chose 222 samples as a supplement to data balance. By SMOTE simulation, we finally have total of 1,151 samples for modeling analysis, of which 269 are abnormal samples, while 882 are normal data from the original. The scattering distribution is shown in <xref ref-type="fig" rid="F4">Figure&#x20;4B</xref>. In this case, we have the sample balance ratio at about 3:1 for the normal samples over the abnormal samples.</p>
<p>Hereafter, the 1,151&#x20;SMOTE-balancing samples were used to train the TSRNN model (defined in <italic>Time-Series RNN Model</italic>), as to build up an intelligent network architecture with adaptive grid optimization of parameters, for accurate recognition of the abnormal power users who are stealing electricity.</p>
</sec>
<sec id="s5">
<title>Discriminations Based on TSRNN Training and Testing</title>
<p>An applicable discrimination model for detecting electricity theft was trained using the TSRNN architecture based on the power consumption data of the 1,151&#x20;SMOTE-balanced samples. The recorded ARP, PEA, and SHO variables are taken as the network input. The data have a time-series record of 820&#xa0;days.</p>
<p>The data samples were divided into two sets for model training and testing: 918 samples (&#x223c;80%) for training and 233 (&#x223c;20%) for testing. The training data were used to conduct the data-driven machine learning optimization of the TSRNN model. The model was constructed with three input neurons and one hidden neuron to produce the output results. There, we have three input-to-hidden linking weights (<inline-formula id="inf53">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>,<inline-formula id="inf54">
<mml:math id="m63">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf55">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) and one hidden-to-output linking weight (<inline-formula id="inf56">
<mml:math id="m65">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula>) to adjust. There is also a linking weight (<inline-formula id="inf57">
<mml:math id="m66">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula>) to help accept another data input from the former time moment of the circle iteration. With machine learning operations, these linking weights were adaptively identified as their most suitable values during the model training process, and then the testing data were used to examine the model discrimination effectiveness by using the data-driven decisive parameters.</p>
<p>In progress, the 918 training samples were introduced to the input layer at every moment of time and then delivered to compute the hidden variables. Notably, the RNN architecture is characterized with the circle of reproducing the hidden layer. The hidden variables at <inline-formula id="inf58">
<mml:math id="m67">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> moment are affected by both the <inline-formula id="inf59">
<mml:math id="m68">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula>-moment input and the hidden variables at the <inline-formula id="inf60">
<mml:math id="m69">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> moment, where <inline-formula id="inf61">
<mml:math id="m70">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>&#x2026;</mml:mo>
<mml:mn>820</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. Thus, a series of phased discriminant results were obtained from the output layers at every time moment. Specifically, we chose to make a segmentation to the full time series from January 1, 2017, to March 31, 2019. There, we set five time markers (see <xref ref-type="table" rid="T2">Table&#x20;2</xref>), to observe five phased modeling outputs for examining the progress of model optimization.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Markers of the five special time nodes for investigation of the TSRNN model performance.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Time marker</th>
<th align="center">Marked moment of time series (<inline-formula id="inf62">
<mml:math id="m71">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula>)</th>
<th align="center">Corresponding time nodes</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<inline-formula id="inf63">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="char" char=".">181</td>
<td align="left">June 30, 2017</td>
</tr>
<tr>
<td align="left">
<inline-formula id="inf64">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="char" char=".">365</td>
<td align="left">December 31, 2017</td>
</tr>
<tr>
<td align="left">
<inline-formula id="inf65">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="char" char=".">546</td>
<td align="left">June 30, 2018</td>
</tr>
<tr>
<td align="left">
<inline-formula id="inf66">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="char" char=".">730</td>
<td align="left">December 31, 2018</td>
</tr>
<tr>
<td align="left">
<inline-formula id="inf67">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>5</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="char" char=".">820</td>
<td align="left">March 31, 2019</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Based on the 918 training samples, the TSRNN model was trained with parameters&#x2019; iteration by circle improvement of the hidden neurons. We calculated the model discriminant indicators at each phase stoppage moment of <inline-formula id="inf68">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf69">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>5</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and drew the ROC curves (see <xref ref-type="fig" rid="F5">Figure&#x20;5</xref>). The ROC figures show that the TSRNN model was continuously improved with the promotion of time series. Eventually, the optimal model was observed at <inline-formula id="inf70">
<mml:math id="m79">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mn>5</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>820</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>ROC curves for the evaluation of the TSRNN training effects at the five selected time markers based on the 918 training samples.</p>
</caption>
<graphic xlink:href="fenrg-09-773805-g005.tif"/>
</fig>
<p>To study the machine learning progress on parameter optimization, we further investigate the running procedures of the adaptive tuning of the TSRNN linking weights. If the linking weights are denoted as a combination of <inline-formula id="inf71">
<mml:math id="m80">
<mml:mrow>
<mml:mo mathvariant="bold">(</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">w</mml:mi>
<mml:mn mathvariant="bold">1</mml:mn>
</mml:msub>
<mml:mo mathvariant="bold">,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">w</mml:mi>
<mml:mn mathvariant="bold">2</mml:mn>
</mml:msub>
<mml:mo mathvariant="bold">,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">w</mml:mi>
<mml:mn mathvariant="bold">3</mml:mn>
</mml:msub>
<mml:mo mathvariant="bold">,</mml:mo>
<mml:mi mathvariant="bold">v</mml:mi>
<mml:mo mathvariant="bold">,</mml:mo>
<mml:mi mathvariant="bold">u</mml:mi>
<mml:mo mathvariant="bold">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, we initialized this combination as <inline-formula id="inf72">
<mml:math id="m81">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>100</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>100</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>100</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>1</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> for model optimization by network iteration of time-series circulation. When time varies, the more and more power consumption data were input to the network, and thus, the linking weights were adjusted for the improving TSRNN model. The changing values of each linking weight were recorded with a time interval of every 20 moments, and thus, we obtained the variation trends of the five linking weights for model optimization (see <xref ref-type="fig" rid="F6">Figure&#x20;6</xref>). It is seen from <xref ref-type="fig" rid="F6">Figure&#x20;6</xref> that the network weights of <inline-formula id="inf73">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf74">
<mml:math id="m83">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula> were presented as an overall downward trend with cyclical recovery fluctuations, ending with optimal values close to zero. And the parameter <inline-formula id="inf75">
<mml:math id="m84">
<mml:mi>u</mml:mi>
</mml:math>
</inline-formula> (i.e.,&#x20;the weight of the iteration of time series) shows a trend of first falling and then rising. In the end, the optimal value of <inline-formula id="inf76">
<mml:math id="m85">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> was recognized as <inline-formula id="inf77">
<mml:math id="m86">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>2.763</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>0.767</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>0.821</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>3.254</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>0.564</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> after 820 iterations by time series, noting that <inline-formula id="inf78">
<mml:math id="m87">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.564</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> was for the circle iterative optimization from <inline-formula id="inf79">
<mml:math id="m88">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>819</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf80">
<mml:math id="m89">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>820</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. These observed optimal values of parameters indicated that the optimal TSRNN model was trained to have a linear formula expression with simple weight coefficients, while the circle iteration of time series pays a certain contribution to the network&#x20;model.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Training of linking weights in the TSRNN structure.</p>
</caption>
<graphic xlink:href="fenrg-09-773805-g006.tif"/>
</fig>
<p>The predictive performance of the TSRNN discriminant model with adaptive tuning of the network weights was further evaluated by the 233 test samples, which were assumed to be &#x201c;unknown&#x201d; because they were not involved in the training process. We have the knowledge that there were 53 abnormal samples and 180 normal samples in the test sample set. The optimal TSRNN model is evaluated with a relative high prediction accuracy upon the quantitative metrics of the model indicators. The predictive ACC, TPR, and FAR were 89.3, 92.5, and 11.7%, respectively. The corresponding confusion matrix is shown in <xref ref-type="table" rid="T3">Table&#x20;3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Confusion matrix of the discriminating results predicted by the optimal TSRNN model for the 233 test samples.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" colspan="2" align="left"/>
<th colspan="2" align="center">Prediction by the optimal TSRNN model</th>
</tr>
<tr>
<th align="left">Positive/abnormal</th>
<th align="left">Negative/normal</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">Actual data</td>
<td align="left">Positive/abnormal</td>
<td align="center">TP &#x3d; 49</td>
<td align="center">FN &#x3d; 4</td>
</tr>
<tr>
<td align="left">Negative/normal</td>
<td align="center">FP &#x3d; 21</td>
<td align="center">TN &#x3d; 159</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Aiming to find out the electricity theft from the real power consumption users, the optimal model output its discriminant results for each sample (shown in <xref ref-type="fig" rid="F7">Figure&#x20;7</xref>). The virtual use data which were produced by SMOTE balancing were not targeted for prediction. Thus, it is necessary to distinguish the real abnormal data from the virtual abnormal data. Practically, we used solid stars to mark the 10 real abnormal samples in the figure, and only two of them were predicted to be false. The results indicated that the adaptive TSRNN architecture is functional to predict the abnormal cases in the daily records of the power consumption&#x20;data.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Discrimination for each test sample by the optimal TSRNN&#x20;model.</p>
</caption>
<graphic xlink:href="fenrg-09-773805-g007.tif"/>
</fig>
<p>Furthermore, the well-trained TSRNN architecture was utilized to monitor the time-series data from January 1, 2017, to March 31, 2019, to recognize the power consumption users who probably have electricity theft behavior. The identification of the real abnormal users is listed in <xref ref-type="table" rid="T4">Table&#x20;4</xref>. It is learnt from <xref ref-type="table" rid="T4">Table&#x20;4</xref> that the optimal TSRNN model successfully identified 44 of the total of 47 abnormal users. The results show that the proposed adaptive TSRNN architecture combined with SMOTE sample balancing technique is able to accurately find the abnormal samples based on the analysis of the time-series&#x2013;recorded power consumption data, thus to recognize the electricity theft behaviors.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Discrimination results for the 47 real abnormal data of the electricity theft&#x20;users.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">User number</th>
<th align="center">Predictions</th>
<th align="center">User number</th>
<th align="center">Predictions</th>
<th align="center">User number</th>
<th align="center">Predictions</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">No. 16</td>
<td align="center">abnormal</td>
<td align="center">No. 331</td>
<td align="center">NORMAL</td>
<td align="center">No. 645</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 24</td>
<td align="center">abnormal</td>
<td align="center">No. 354</td>
<td align="center">abnormal</td>
<td align="center">No. 653</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 46</td>
<td align="center">abnormal</td>
<td align="center">No. 371</td>
<td align="center">abnormal</td>
<td align="center">No. 669</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 49</td>
<td align="center">abnormal</td>
<td align="center">No. 375</td>
<td align="center">abnormal</td>
<td align="center">No. 690</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 62</td>
<td align="center">abnormal</td>
<td align="center">No. 397</td>
<td align="center">abnormal</td>
<td align="center">No. 694</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 76</td>
<td align="center">abnormal</td>
<td align="center">No. 415</td>
<td align="center">abnormal</td>
<td align="center">No. 696</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 99</td>
<td align="center">abnormal</td>
<td align="center">No. 462</td>
<td align="center">abnormal</td>
<td align="center">No. 706</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 114</td>
<td align="center">abnormal</td>
<td align="center">No. 482</td>
<td align="center">abnormal</td>
<td align="center">No. 726</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 138</td>
<td align="center">abnormal</td>
<td align="center">No. 502</td>
<td align="center">abnormal</td>
<td align="center">No. 737</td>
<td align="center">NORMAL</td>
</tr>
<tr>
<td align="left">No. 152</td>
<td align="center">abnormal</td>
<td align="center">No. 509</td>
<td align="center">abnormal</td>
<td align="center">No. 765</td>
<td align="center">NORMAL</td>
</tr>
<tr>
<td align="left">No. 226</td>
<td align="center">abnormal</td>
<td align="center">No. 609</td>
<td align="center">abnormal</td>
<td align="center">No. 773</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 262</td>
<td align="center">abnormal</td>
<td align="center">No. 613</td>
<td align="center">abnormal</td>
<td align="center">No. 812</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 263</td>
<td align="center">abnormal</td>
<td align="center">No. 614</td>
<td align="center">abnormal</td>
<td align="center">No. 817</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 264</td>
<td align="center">abnormal</td>
<td align="center">No. 636</td>
<td align="center">abnormal</td>
<td align="center">No. 833</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 281</td>
<td align="center">abnormal</td>
<td align="center">No. 637</td>
<td align="center">abnormal</td>
<td align="center">No. 864</td>
<td align="center">abnormal</td>
</tr>
<tr>
<td align="left">No. 286</td>
<td align="center">abnormal</td>
<td align="center">No. 638</td>
<td align="center">abnormal</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec sec-type="conclusion" id="s6">
<title>Conclusion</title>
<p>In this paper, an adaptive TSRNN architecture was built up to detect the electricity theft based on time-series data of the power consumption. The recorded data were monitored continuously from January 1, 2017, to March 31, 2019 (820&#xa0;days in total). By monitoring the ARP, PEA, and SHO data, the users who are suspicious of stealing electricity were denoted as abnormal samples, while the other common users were denoted as normal. There, we had collected the data of 882 normal samples and 47 abnormal samples. As the abnormal users appear as the minority in all of the recorded data, the SMOTE algorithm was used to ease the data imbalance by generating 222 virtual abnormal samples, to make the ratio of the normal over the abnormal at about&#x20;3:1.</p>
<p>The TSRNN model was established based on the total of 1,151 user samples over the 820&#x20;time-series moments. A basic network was formed with three input nodes for receiving the data in the three variables of ARP, PEA, and SHO, and with one hidden neuron for extracting data features. Then, the network output was computed as a k-means classified result to discriminate the sample as an abnormal one or a normal one. The k-means classifier calculation was on the basis of Mahalanobis distance. As for the successive analysis of the non-stopping input time-series data, the TSRNN structure was re-formed by circulating this kind of basic network. Then, each hidden node was influenced by the input data at the current time moment and the data delivery from the time-former hidden node, and thus, the output results can be optimized by adaptively tuning the network parameters in the combination of linking weights <inline-formula id="inf81">
<mml:math id="m90">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>u</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. In our empirical experiment, the most optimal values of the combination of linking weights were observed as <inline-formula id="inf82">
<mml:math id="m91">
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>2.763</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>0.767</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>0.821</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>3.254</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mn>0.564</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> after 820 iterations by time series. There, we obtained the discriminant model with a high prediction accuracy of ACC &#x3d; 95.1%. The optimal TSRNN model was evaluated to be much effective by the 233 test samples, with the testing ACC &#x3d; 89.3, TPR &#x3d; 92.5, and FAR &#x3d; 11.7%. Therefore, the adaptive TSRNN model was finally used to predict the 47 real abnormal samples, and the discriminating results are quite appreciating, with only three samples predicted to be false. The prediction accuracy was as high as&#x20;93.6%.</p>
<p>The experimental results indicated that the proposed adaptive TSRNN architecture in fusion with the SMOTE balancing technique is feasible to extract data features for monitoring the abnormal electricity theft behavior. The methodology framework is prospectively promoted to be used for online monitoring on big&#x20;data analysis for a large scale of electricity power consumption.</p>
</sec>
</body>
<back>
<sec id="s7">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material, and further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s8">
<title>Author Contributions</title>
<p>YL conceptualized the idea and supervised the work. GL and SH performed the methodology. GL and HW visualized the results. HW was involved in formal analysis. SH and ZN investigated the data. ZN validated the data.GL and HF wrote the original draft. HF curated the data and ran the software. XF and SH reviewed and edited the paper. XF obtained the resources.</p>
</sec>
<sec id="s9">
<title>Funding</title>
<p>This research was funded by the project supported by the China Southern Power Grid Corporation (Grant No. GDKJXM20185800).</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of Interest</title>
<p>The author HW was employed by Zhanjiang Power Supply Bureau of Guangdong Power Grid Co.,&#x20;Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmad</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hasan</surname>
<given-names>D. Q. U.</given-names>
</name>
<name>
<surname>Zada</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Non-Technical Loss Detection, Prevention and Suppression Issues for AMI in Smart Grid</article-title>. <source>Ijser</source> <volume>6</volume> (<issue>3</issue>), <fpage>217</fpage>&#x2013;<lpage>228</lpage>. <pub-id pub-id-type="doi">10.14299/ijser.2015.03.001</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Dahidi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ayadi</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Adeeb</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Louzazni</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Assessment of Artificial Neural Networks Learning Algorithms and Training Datasets for Solar Photovoltaic Power Production Prediction</article-title>. <source>Front. Energ. Res.</source> <volume>7</volume>, <fpage>1</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.3389/fenrg.2019.00130</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alkinani</surname>
<given-names>H. H.</given-names>
</name>
<name>
<surname>Al-Hameedi</surname>
<given-names>A. T. T.</given-names>
</name>
<name>
<surname>Dunn-Norman</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Data-driven Recurrent Neural Network Model to Predict the Rate of Penetration</article-title>. <source>Upstream Oil Gas Techn.</source> <volume>7</volume>, <fpage>100047</fpage>. <pub-id pub-id-type="doi">10.1016/j.upstre.2021.100047</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aryanezhad</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A Novel Approach to Detection and Prevention of Electricity Pilferage over Power Distribution Network</article-title>. <source>Int. J.&#x20;Electr. Power Energ. Syst.</source> <volume>111</volume>, <fpage>191</fpage>&#x2013;<lpage>200</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2019.04.005</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Avila</surname>
<given-names>N. F.</given-names>
</name>
<name>
<surname>Figueroa</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Chu</surname>
<given-names>C.-C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>NTL Detection in Electric Distribution Systems Using the Maximal Overlap Discrete Wavelet-Packet Transform and Random Undersampling Boosting</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>33</volume>, <fpage>7171</fpage>&#x2013;<lpage>7180</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2018.2853162</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Detection of Power Theft Behavior of Distribution Network Based on RBF Neural Network</article-title>. <source>J.&#x20;Yunnan Univ. Nat. Sci. Ed.</source> <volume>40</volume> (<issue>5</issue>), <fpage>872</fpage>&#x2013;<lpage>878</lpage>. <pub-id pub-id-type="doi">10.7540/j.ynu.20170426</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chawla</surname>
<given-names>N. V.</given-names>
</name>
<name>
<surname>Bowyer</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>L. O.</given-names>
</name>
<name>
<surname>Kegelmeyer</surname>
<given-names>W. P.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>SMOTE: Synthetic Minority Over-sampling Technique</article-title>. <source>jair</source> <volume>16</volume>, <fpage>321</fpage>&#x2013;<lpage>357</lpage>. <pub-id pub-id-type="doi">10.1613/jair.953</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A Combination Strategy of Random forest and Back Propagation Network for Variable Selection in Spectral Calibration</article-title>. <source>Chemometrics Intell. Lab. Syst.</source> <volume>182</volume>, <fpage>101</fpage>&#x2013;<lpage>108</lpage>. <pub-id pub-id-type="doi">10.1016/j.chemolab.2018.09.002</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Mo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A Hybrid Optimization Method for Sample Partitioning in Near-Infrared Analysis</article-title>. <source>Spectrochimica Acta A: Mol. Biomol. Spectrosc.</source> <volume>248</volume>, <fpage>119182</fpage>. <pub-id pub-id-type="doi">10.1016/j.saa.2020.119182</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cossu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Carta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lomonaco</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Bacciu</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Continual Learning for Recurrent Neural Networks: An Empirical Evaluation</article-title>. <source>Neural Networks</source> <volume>143</volume>, <fpage>607</fpage>&#x2013;<lpage>627</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2021.07.021</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dileep</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A Survey on Smart Grid Technologies and Applications</article-title>. <source>Renew. Energ.</source> <volume>146</volume>, <fpage>2589</fpage>&#x2013;<lpage>2625</lpage>. <pub-id pub-id-type="doi">10.1016/j.renene.2019.08.092</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Research on Electricity Anti-stealing Method Based on Power Consumption Information Acquisition and Big Data</article-title>. <source>Elec. Meas. Instrum.</source> <volume>55</volume> (<issue>21</issue>), <fpage>43</fpage>&#x2013;<lpage>49</lpage>. </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Farjaminezhad</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Safari</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Moghadam</surname>
<given-names>A. M. E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Recurrent Neural Networks Models for Analyzing Single and Multiple Transient Faults in Combinational Circuits</article-title>. <source>Microelectronics J.</source> <volume>112</volume>, <fpage>104993</fpage>. <pub-id pub-id-type="doi">10.1016/j.mejo.2021.104993</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fern&#xe1;ndez</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Garc&#xed;a</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Herrera</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chawla</surname>
<given-names>N. V.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>SMOTE for Learning from Imbalanced Data: Progress and Challenges, Marking the 15-year Anniversary</article-title>. <source>jair</source> <volume>61</volume>, <fpage>863</fpage>&#x2013;<lpage>905</lpage>. <pub-id pub-id-type="doi">10.1613/jair.1.11192</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Garcia</surname>
<given-names>E. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Learning from Imbalanced Data</article-title>. <source>IEEE Trans. Knowl. Data Eng.</source> <volume>21</volume>, <fpage>1263</fpage>&#x2013;<lpage>1284</lpage>. <pub-id pub-id-type="doi">10.1109/tkde.2008.239</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Nontechnical Loss Detection Based on Stacked Uncorrelating Autoencoder and Support Vector Machine</article-title>. <source>Autom. Elec. Power Syst.</source> <volume>43</volume> (<issue>1</issue>), <fpage>119</fpage>&#x2013;<lpage>127</lpage>. <pub-id pub-id-type="doi">10.7500/AEPS20180630013</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Data Normalization to Accelerate Training for Linear Neural Net to Predict Tropical Cyclone Tracks</article-title>. <source>Math. Probl. Eng.</source> <volume>2015</volume>. <pub-id pub-id-type="doi">10.1155/2015/931629</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jokar</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Arianpoo</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Leung</surname>
<given-names>V. C. M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Electricity Theft Detection in AMI Using Customers&#x27; Consumption Patterns</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>7</volume>, <fpage>216</fpage>&#x2013;<lpage>226</lpage>. <pub-id pub-id-type="doi">10.1109/tsg.2015.2425222</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yingchen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Electricity Theft Detection in Power Grids with Deep Learning and Random Forests</article-title>. <source>J.&#x20;Electr. Comput. Eng.</source> <volume>2019</volume>, <fpage>4136874</fpage>. <pub-id pub-id-type="doi">10.1155/2019/4136874</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A Recurrent Neural Network Framework with an Adaptive Training Strategy for Long-Time Predictive Modeling of Nonlinear Dynamical Systems</article-title>. <source>J.&#x20;Sound Vibration</source> <volume>506</volume>, <fpage>116167</fpage>. <pub-id pub-id-type="doi">10.1016/j.jsv.2021.116167</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ni</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021a</year>). <article-title>Many-objective Distribution Network Reconfiguration via Deep Reinforcement Learning Assisted Optimization Algorithm</article-title>. <source>IEEE Trans. Power Deliv.</source>, <fpage>1</fpage>. <pub-id pub-id-type="doi">10.1109/tpwrd.2021.3107534</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>The Intelligent Analysis on the Trend Anomaly of the Electric Energy Meter Based on LOF Algorithm</article-title>. <source>Elec. Meas. Instrum.</source> <volume>53</volume> (<issue>18</issue>), <fpage>69</fpage>&#x2013;<lpage>73</lpage>. </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gooi</surname>
<given-names>H. B.</given-names>
</name>
</person-group> (<year>2021b</year>). <article-title>Deep Learning Based Densely Connected Network for Load Forecasting</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>36</volume>, <fpage>2829</fpage>&#x2013;<lpage>2840</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2020.3048359</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Finch</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Utiyama</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sumita</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Agreement on Target-Bidirectional Recurrent Neural Networks for Sequence-To-Sequence Learning</article-title>. <source>jair</source> <volume>67</volume>, <fpage>581</fpage>&#x2013;<lpage>606</lpage>. <pub-id pub-id-type="doi">10.1613/jair.1.12008</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mozaffar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Paul</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Al-Bahrani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wolff</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Choudhary</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Data-driven Prediction of the High-Dimensional thermal History in Directed Energy Deposition Processes via Recurrent Neural Networks</article-title>. <source>Manufacturing Lett.</source> <volume>18</volume>, <fpage>35</fpage>&#x2013;<lpage>39</lpage>. <pub-id pub-id-type="doi">10.1016/j.mfglet.2018.10.002</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>Z. J.</given-names>
</name>
<name>
<surname>Vyakaranam</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Etingov</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Power System Event Classification and Localization Using a Convolutional Neural Network</article-title>. <source>Front. Energ. Res.</source> <volume>8</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.3389/fenrg.2020.607826</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>St&#xe5;hl</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Mathiason</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Falkman</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Karlsson</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Using Recurrent Neural Networks with Attention for Detecting Problematic Slab Shapes in Steel Rolling</article-title>. <source>Appl. Math. Model.</source> <volume>70</volume>, <fpage>365</fpage>&#x2013;<lpage>377</lpage>. <pub-id pub-id-type="doi">10.1016/j.apm.2019.01.027</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Aziz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Voropai</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Solar Irradiance Forecasting Based on Direct Explainable Neural Network</article-title>. <source>Energ. Convers. Manag.</source> <volume>226</volume>, <fpage>113487</fpage>. <pub-id pub-id-type="doi">10.1016/j.enconman.2020.113487</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Electricity Theft Detection for Customers in Power Utility Based on Real-Valued Deep Belief Network</article-title>. <source>Power Syst. Techn.</source> <volume>43</volume> (<issue>3</issue>), <fpage>1083</fpage>&#x2013;<lpage>1091</lpage>. </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Or</surname>
<given-names>S. W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Voropai</surname>
<given-names>N. I.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Optimal Coordinated Control of Multi-Renewable-To-Hydrogen Production System for Hydrogen Fueling Stations</article-title>. <source>IEEE Trans. Ind. Applicat.</source>, <fpage>1</fpage>. <pub-id pub-id-type="doi">10.1109/TIA.2021.3093841</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ai</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Energy Theft Detection in an Edge Data center Using Threshold-Based Abnormality Detector</article-title>. <source>Int. J.&#x20;Electr. Power Energ. Syst.</source> <volume>121</volume>, <fpage>106162</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2020.106162</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>Z. Y.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Imbalance Learning Machine-Based Power System Short-Term Voltage Stability Assessment</article-title>. <source>IEEE Trans. Ind. Inf.</source> <volume>13</volume>, <fpage>2533</fpage>&#x2013;<lpage>2543</lpage>. <pub-id pub-id-type="doi">10.1109/tii.2017.2696534</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Distributed Clustering Algorithm for Awareness of Electricity Consumption Characteristics of Massive Consumers</article-title>. <source>Autom. Elec. Power Syst.</source> <volume>40</volume> (<issue>12</issue>), <fpage>21</fpage>&#x2013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.7500/AEPS20160316007</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>