<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Energy Res.</journal-id>
<journal-title>Frontiers in Energy Research</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Energy Res.</abbrev-journal-title>
<issn pub-type="epub">2296-598X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1269854</article-id>
<article-id pub-id-type="doi">10.3389/fenrg.2023.1269854</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Energy Research</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Combination optimization method of grid sections based on deep reinforcement learning with accelerated convergence speed</article-title>
<alt-title alt-title-type="left-running-head">Zhao et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenrg.2023.1269854">10.3389/fenrg.2023.1269854</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Zhao</surname>
<given-names>Huashi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Wu</surname>
<given-names>Zhichao</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="fn" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>He</surname>
<given-names>Yubin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Fu</surname>
<given-names>Qiujia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liang</surname>
<given-names>Shouyu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ma</surname>
<given-names>Guang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Wenchao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yang</surname>
<given-names>Qun</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2394185/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>China Southern Power Grid Dispatching and Control Center</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>College of Computer Science and Technology/College of Artificial Intelligence/College of Software, Nanjing University of Aeronautics and Astronautics</institution>, <addr-line>Nanjing</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1490679/overview">Chixin Xiao</ext-link>, University of Wollongong, Australia</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1312399/overview">Linfei Yin</ext-link>, Guangxi University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2087583/overview">Huifeng Zhang</ext-link>, Nanjing University of Posts and Telecommunications, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Qun Yang, <email>qun.yang@nuaa.edu.cn</email>
</corresp>
<fn fn-type="equal" id="fn001">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work and share first authorship</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>06</day>
<month>10</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>11</volume>
<elocation-id>1269854</elocation-id>
<history>
<date date-type="received">
<day>31</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>09</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Zhao, Wu, He, Fu, Liang, Ma, Li and Yang.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Zhao, Wu, He, Fu, Liang, Ma, Li and Yang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>A modern power system integrates more and more new energy and uses a large number of power electronic equipment, which makes it face more challenges in online optimization and real-time control. Deep reinforcement learning (DRL) has the ability of processing big data and high-dimensional features, as well as the ability of independently learning and optimizing decision-making in complex environments. This paper explores a DRL-based online combination optimization method of grid sections for a large complex power system. In order to improve the convergence speed of the model, it proposes to discretize the output action of the unit and simplify the action space. It also designs a reinforcement learning loss function with strong constraints to further improve the convergence speed of the model and facilitate the algorithm to obtain a stable solution. Moreover, to avoid the local optimal solution problem caused by the discretization of the output action, this paper proposes to use the annealing optimization algorithm to make the granularity of the unit output finer. The proposed method in this paper has been verified on an IEEE 118-bus system. The experimental results show that it has fast convergence speed and better performance and can obtain stable solutions.</p>
</abstract>
<kwd-group>
<kwd>grid section</kwd>
<kwd>deep reinforcement learning</kwd>
<kwd>convergence speed</kwd>
<kwd>discretize</kwd>
<kwd>loss function</kwd>
<kwd>annealing optimization algorithm</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Smart Grids</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The fundamental issue of power systems is to ensure that the grid operates economically, reliably, and stably. At present, as new energy develops rapidly and its proportion in the total power supply continues to increase, power systems face new challenges in terms of real-time dispatch and stability control.</p>
<p>Most of the traditional power dispatching solutions are based on accurate modeling of the system, mainly using classical methods, metaheuristic methods, and hybrid methods. In order to solve the constrained economic dispatch problem, <xref ref-type="bibr" rid="B6">Gherbi and Lakdja (2011)</xref> proposed a quadratic programming method based on a variable transformation technique to handle the linearization of constraints. <xref ref-type="bibr" rid="B8">Irisarri et al. (1998)</xref> studied the interior point method, which is one of the methods for dealing with constrained optimization problems. <xref ref-type="bibr" rid="B22">Zhan et al. (2013)</xref> investigated a fast iteration method. Different from theseclassical methods, <xref ref-type="bibr" rid="B10">Larouci et al. (2022)</xref> improved four metaheuristic algorithms, while <xref ref-type="bibr" rid="B14">Modiri-Delshad et al. (2016)</xref> presented a new backtracking search algorithm that utilizes crossover and mutation operators to efficiently explore search domains. Among the hybrid methods, <xref ref-type="bibr" rid="B3">Ayd&#x131;n and &#xd6;zy&#xf6;n (2013)</xref> used incremental artificial bee colony (IABC) algorithm, together with local search, to solve the non-convex economic dispatch problem, whereas <xref ref-type="bibr" rid="B1">Alshammari et al. (2022)</xref> extended IABC and introduced four various chaotic maps in all phases of the artificial bee colony algorithm to generate the random variables.</p>
<p>However, in modern power systems, the integration of renewable energy brings more randomness into the energy output of unit commitment. It greatly increases the uncertainty of system operation while decreasing the system&#x2019;s ability to resist faults. In modern power systems, traditional power dispatching methods face several problems, such as large action space, long decision-making steps, high computational complexity, and poor performance. They also have to deal with uncertainty and sudden situations.</p>
<p>Power dispatching is a multi-constraint, nonlinear, and high-dimensional optimization decision problem. Recently, deep learning (DL) has been applied to the optimization and control of smart grids as it has the powerful feature representation ability, as well as the approximation function of neural networks (<xref ref-type="bibr" rid="B21">Yin et al., 2018</xref>; <xref ref-type="bibr" rid="B2">Ardakani and Bouffard, 2018</xref>; <xref ref-type="bibr" rid="B5">Diehl, 2019</xref>). On the other hand, reinforcement learning (RL) algorithms, such as Q-learning, SARSA, distributional RL, policy gradient, DDPG, and A3C, have also been adopted in modern power grids. Furthermore, deep reinforcement learning (DRL) combines the decision-making ability of RL and the ability of processing large data and high-dimensional features of DL, which makes it very suitable for power dispatching.</p>
<p>The basic principle of RL is that the agent performs a series of actions in an environment and obtains feedback from the environment to adjust its strategy, thus achieving optimal decision-making. <xref ref-type="bibr" rid="B20">Yan and Xu (2020)</xref> proposed an optimal power flow method based on Lagrangian deep reinforcement learning for real-time optimization of power grid control. <xref ref-type="bibr" rid="B7">Guo et al. (2022)</xref> implemented online AC-OPF by combining reinforcement learning and imitation learning. Imitation learning is introduced to improve the learning efficiency of agents in reinforcement learning by learning from expert experience. <xref ref-type="bibr" rid="B9">Jiang et al. (2021)</xref> used a deep Q-network (DQN) to model the reactive voltage optimization problem. <xref ref-type="bibr" rid="B23">Zhao et al. (2022)</xref> and <xref ref-type="bibr" rid="B24">Zhou et al. (2021)</xref> used the policy-based reinforcement learning algorithm PPO to realize autonomous dispatching of the power system. Different from <xref ref-type="bibr" rid="B24">Zhou et al. (2021)</xref>, <xref ref-type="bibr" rid="B23">Zhao et al. (2022)</xref> combined the graph neural network (GNN) with reinforcement learning to model the power grid structure and its topological changes, achieving autonomous dispatch of the power system with variable topology. <xref ref-type="bibr" rid="B12">Liu et al. (2022)</xref> explored how to autonomously control the power system under the influence of extreme weather. They proposed a DRL method based on imitation learning. The imitation learning module interacts with agents during reinforcement learning, making the system operate as much as possible in the original topology. <xref ref-type="bibr" rid="B15">Sayed et al. (2022)</xref> aimed at the AC-OPF problem. They proposed a DRL method based on the penalty convex process. A systematic control strategy is obtained through DRL, and the operation constraint is satisfied by using the convex safety layer.</p>
<p>All of the aforementioned works have investigated the application of deep reinforcement learning in power dispatching. This paper explores the section control of the modern power system integrated with new energy. It aims at online optimization for a large-scale power system whose optimization goals are complex. In this paper, a DRL method with accelerated convergence speed is proposed to solve the problem of dimensional disaster that occurs when the problem scale and decision variables increase. The proposed method also addresses the problem of the dispatching algorithm where it is difficult to obtain a stable solution because the optimization targets are coupling and mutually constrained, and moreover, each target has inconsistent sensitivity to the unit adjustment.</p>
<p>The contributions of this paper are as follows:<list list-type="simple">
<list-item>
<p>(1) The paper proposes a combination optimization method for grid dispatching based on deep reinforcement learning in which it simplifies the action space and improves the convergence speed of the model by discretizing the unit output action.</p>
</list-item>
<list-item>
<p>(2) A reinforcement learning loss function with strong constraints is proposed to further improve the convergence speed of the model as well as achieve the stability of the algorithm solution.</p>
</list-item>
<list-item>
<p>(3) The annealing optimization algorithm is proposed to make the granularity of the unit output finer and avoid the problem of local optimal solutions caused by the discretization of the output actions.</p>
</list-item>
</list>
</p>
<p>Experimental results on an IEEE 118-bus system show that the method proposed in this paper is effective. By using the proposed method, the convergence speed of the DRL model is faster, and stable solutions can be achieved.</p>
</sec>
<sec id="s2">
<title>2 Mathematical model for combination optimization of grid sections</title>
<sec id="s2-1">
<title>2.1 Objective function</title>
<p>With the objective of minimizing the total power generation costs of the hydro, thermal, and wind power multi-energy complementary systems and improving the system&#x2019;s new energy consumption (see Eq. <xref ref-type="disp-formula" rid="e1">1</xref> for details), a short-term optimal scheduling model for combination optimization of grid sections is established.<disp-formula id="e1">
<mml:math id="m4">
<mml:mtable class="align" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="italic">min</mml:mi>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:mfenced open="(" close="">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mspace width="2em"/>
<mml:mspace width="2em"/>
<mml:mspace width="1em"/>
<mml:mfenced open="" close=")">
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>C</italic>
<sub>
<italic>i</italic>,<italic>t</italic>
</sub>(<italic>p</italic>
<sub>
<italic>i</italic>,<italic>t</italic>
</sub>) is the operating cost of the <italic>i</italic>th generator unit at interval <italic>t</italic>. It is a quadratic function (<xref ref-type="bibr" rid="B25">Zivic Djurovic et al., 2012</xref>) of the unit&#x2019;s output interval and the corresponding energy price (see Eq. <xref ref-type="disp-formula" rid="e2">2</xref> for details). <italic>p</italic>
<sub>
<italic>i</italic>,<italic>t</italic>
</sub> is the active power output of the <italic>i</italic>th generator unit at time <italic>t</italic>; <italic>w</italic>
<sub>1</sub>, <italic>w</italic>
<sub>2</sub>, <italic>w</italic>
<sub>3</sub>, and <italic>w</italic>
<sub>4</sub> are combination coefficients; <italic>I</italic>
<sub>
<italic>t</italic>
</sub> is the thermal generator sets; <italic>I</italic>
<sub>
<italic>w</italic>
</sub> is the hydroelectric generator sets; and <italic>I</italic>
<sub>
<italic>ne</italic>
</sub> is the wind and solar power generator sets. <inline-formula id="inf4">
<mml:math id="m5">
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is the maximum active output of the <italic>i</italic>th generator unit; <italic>T</italic> is the number of time slots in the scheduling cycle; and <italic>N</italic> is the number of units participating in the combination calculation.<disp-formula id="e2">
<mml:math id="m6">
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2a;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>&#x2a;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
</mml:math>
<label>(2)</label>
</disp-formula>where <italic>a</italic>, <italic>b</italic>, and <italic>c</italic> are the coefficients for the quadratic, linear, and constant terms of the operating cost function, respectively.</p>
</sec>
<sec id="s2-2">
<title>2.2 Constraints</title>
<p>
<list list-type="simple">
<list-item>
<p>(1) Load balance constraint.</p>
</list-item>
</list>
</p>
<p>In the power system, the total output of the generator units should be equal to the system load at any time, and this can be expressed as follows:<disp-formula id="e3">
<mml:math id="m8">
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2200;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:math>
<label>(3)</label>
</disp-formula>where <italic>L</italic>
<sub>
<italic>t</italic>
</sub> is the total load data of the power system at time <italic>t</italic>.<list list-type="simple">
<list-item>
<p>(2) Maximum and minimum output constraints of generator units.</p>
</list-item>
</list>
</p>
<p>Considering the generator unit&#x2019;s physical properties (<xref ref-type="bibr" rid="B16">Shchetinin et al., 2018</xref>), its output is adjustable within a certain range, and this can be expressed as follows:<disp-formula id="e4">
<mml:math id="m10">
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf7">
<mml:math id="m11">
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> and <inline-formula id="inf8">
<mml:math id="m12">
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> are the minimum and maximum output of the <italic>n</italic>th generator unit, respectively.<list list-type="simple">
<list-item>
<p>(3) Cross-section power flow limit constraint</p>
</list-item>
</list>
</p>
<p>In the power system, the active power flow of the grid section should be within a certain range at any time (<xref ref-type="bibr" rid="B4">Bakirtzis et al., 2002</xref>), and this can be expressed as<disp-formula id="e5">
<mml:math id="m14">
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2264;</mml:mo>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>,</mml:mo>
</mml:math>
<label>(5)</label>
</disp-formula>where <italic>P</italic>
<sub>
<italic>s</italic>
</sub>(<italic>a</italic>) is the active power flow of the section <italic>s</italic> based on the current output <italic>p</italic> of the generator unit, <inline-formula id="inf10">
<mml:math id="m15">
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is the active power flow limit of the section <italic>s</italic>, and <italic>S</italic> represents the section set.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Combination optimization of grid sections</title>
<sec id="s3-1">
<title>3.1 Deep reinforcement learning</title>
<p>Reinforcement learning is an important method for solving optimization problems. Its mathematical basis is the Markov decision process (MDP). The components of MDP include state space, action space, state transition function, and reward function. Reinforcement learning implements MDP with agent, environment, state, reward, and action.</p>
<p>Most recent works have combined deep learning with reinforcement learning, which is called DRL. In DRL, deep learning models are used to learn the value function or the policy function so that agents can learn to make better decisions. Commonly used DRL algorithms include DQN (deep Q-network) (<xref ref-type="bibr" rid="B13">Mnih et al., 2013</xref>), DDPG (deep deterministic policy gradient) (<xref ref-type="bibr" rid="B11">Lillicrap et al., 2015</xref>), and actor&#x2013;critic (<xref ref-type="bibr" rid="B19">Sutton et al., 1999</xref>).</p>
<p>This paper adopts the actor&#x2013;critic (AC) algorithm and introduces two neural networks into it. One is the policy network, and the other is the value network.</p>
<p>The policy network <italic>&#x3c0;</italic>(<italic>a</italic>&#x7c;<italic>s</italic>; <italic>&#x3b8;</italic>) is equivalent to an actor. It chooses the action <italic>a</italic> based on the state <italic>s</italic>, which is fed back by the environment. The value network plays the role of a critic. It evaluates the policy by using the value network <italic>q</italic> (<italic>s</italic>; <italic>v</italic>). <italic>&#x3b8;</italic> and <italic>v</italic> are the parameters to be trained in the policy network and value network, respectively.</p>
<p>The objective of the policy network is to obtain a higher evaluation by adjusting the action. The policy network in the AC algorithm adopts a policy gradient (PG) network to optimize the policy. In the optimization method, the agent learns to estimate the expected reward of each state and uses the learned knowledge to decide how to choose the action.</p>
<p>The value network evaluates the action of the policy network and feeds back a temporal difference (TD) (<xref ref-type="bibr" rid="B18">Sutton, 1988</xref>) value to the policy network, determining whether the behavior of the policy network is good or bad.</p>
<p>Although the basic AC algorithm is a good idea, it needs to be improved due to the difficulty in convergence. A DRL method with accelerated convergence speed is proposed in this paper. The overall structure of the proposed method is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. In addition, in order to reduce the network update error, the TD error (<xref ref-type="bibr" rid="B17">Silver et al., 2014</xref>) with baseline is incorporated. Moreover, the asynchronous parallel computing method is also used in order to maximize the computing performance.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Overall structure of the method.</p>
</caption>
<graphic xlink:href="fenrg-11-1269854-g001.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Environment setting for reinforcement learning</title>
<p>The basic elements of this reinforcement learning environment are as follows:<list list-type="simple">
<list-item>
<p>(1) Environment. The environment mainly includes various grid section information, such as grid topology, system load, bus load, generator unit status, and section data. In addition, grid system constraints exist in the environment, including power flow constraints, load balance constraints, and generator unit constraints.</p>
</list-item>
<list-item>
<p>(2) Agents. It is a set of generator units participating in the combination optimization calculation of grid sections.</p>
</list-item>
<list-item>
<p>(3) State space. The state space includes current active power output of generator units, system load, bus load, and branch load. The state transition function refers to the probability that the generator unit will take the next action in the current state.</p>
</list-item>
<list-item>
<p>(4) Actions and action space. Actions represent current decisions made by the agent. Action space represents the set of all possible decisions. In the combination optimization problem of the grid section, action represents the active power output of the generator unit at the next moment. Action space is all possible values of the active power output of generators, which is constrained by the maximum and minimum values of the generator&#x2019;s output.</p>
</list-item>
</list>
</p>
<p>In order to improve the learning speed of the policy network, this paper simplifies the action space from the absolute output of the generator unit to one of the three discrete values, namely, 1, &#x2212;1, or 0, which represent that the next output of the generator unit is upward-adjusted (represented as 1 in <xref ref-type="table" rid="T1">Table 1</xref>), downward-adjusted (&#x2212;1), or not adjusted (0), respectively. This optimization method transforms the multi-dimensional continuous action space into a multi-dimensional discrete action space, avoiding the curse of dimensionality and slow model convergence (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Action space.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Generator</th>
<th align="left">Traditional method action space</th>
<th align="left">Proposed method action space</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">0</td>
<td align="left">[0,30]</td>
<td align="left">{&#x2212;1, 0, 1}</td>
</tr>
<tr>
<td align="left">1</td>
<td align="left">[0,100]</td>
<td align="left">{&#x2212;1, 0, 1}</td>
</tr>
<tr>
<td align="left">&#x22ee;</td>
<td align="left">&#x22ee;</td>
<td align="left">&#x22ee;</td>
</tr>
<tr>
<td align="left">N</td>
<td align="left">[0,80]</td>
<td align="left">{&#x2212;1, 0, 1}</td>
</tr>
<tr>
<td align="left">Action space</td>
<td align="left">
<italic>&#x221e;</italic>
</td>
<td align="left">3<sup>
<italic>N</italic>
</sup>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<list list-type="simple">
<list-item>
<p>(5) Reward function. The reward function represents the reward value obtained by the agent after taking a certain action. The optimization goal is to obtain the maximum reward value. In view of the combination optimization problem of the grid section, this paper designs five types of rewards: 1) system cost rewards, 2) power flow limitation rewards, 3) load balancing rewards, 4) clean energy consumption rewards, and 5) generator unit limitation rewards. The purpose of optimizing the reward function is to minimize the system cost and maximize the proportion of clean energy on the premise that the power flow does not exceed the boundary, the output of the generator unit does not exceed the boundary, and the load is balanced in the grid system.</p>
</list-item>
</list>
</p>
<p>For each time step t, the evaluation score <italic>R</italic>
<sub>
<italic>t</italic>
</sub> of the system is calculated as follows:<disp-formula id="e6">
<mml:math id="m21">
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:math>
<label>(6)</label>
</disp-formula>where <italic>r</italic>
<sub>
<italic>i</italic>,<italic>t</italic>
</sub> is the reward of the <italic>i</italic>th type at the time step <italic>t</italic>. For simplicity, the subscript in the following formulas is omitted. Specifically, the calculation of each type of reward is as follows:<list list-type="simple">
<list-item>
<p>1) System cost (positive reward) with the value range of <italic>A</italic>
<sub>0</sub>&#x2a;[0,100]:</p>
</list-item>
</list>
<disp-formula id="e7">
<mml:math id="m22">
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>&#x2a;</mml:mo>
<mml:mi>min</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(7)</label>
</disp-formula>where <italic>A</italic>
<sub>0</sub> &#x3d; 1 is the score weight. <italic>c</italic>
<sub>
<italic>i</italic>
</sub> is the cost of the corresponding generator, and the system has N generators in total. <italic>C</italic>
<sub>
<italic>min</italic>
</sub> is the normalization constant, which is the minimum cost of the system at a moment over a period of time. The lower the system cost is, the higher the reward score is.<list list-type="simple">
<list-item>
<p>2) Power flow limit reward (positive reward) with the value range of <italic>A</italic>
<sub>1</sub>&#x2a;[0, 100]:</p>
</list-item>
</list>
<disp-formula id="e8">
<mml:math id="m23">
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>100</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mspace width="1em"/>
<mml:mo>,</mml:mo>
</mml:math>
<label>(8)</label>
</disp-formula>where <italic>A</italic>
<sub>1</sub> &#x3d; 4 is the score weight, <italic>S</italic> is the total number of sections, and <italic>r</italic>
<sup>
<italic>s</italic>
</sup> is the reward value of the <italic>s</italic>th section. It is calculated according to different situations (over-limit or normal). In over-limit situations (that is, exceeding the upper or lower limit of the predetermined value), severe penalties are imposed, whereas under normal circumstances, there is no penalty. The specific calculation method is as follows:<disp-formula id="e9">
<mml:math id="m24">
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="array">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(9)</label>
</disp-formula>where <italic>P</italic>
<sub>
<italic>s</italic>
</sub> is the power flow of section <italic>s</italic>, <inline-formula id="inf16">
<mml:math id="m25">
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is the upper limit of section <italic>s</italic>, and <inline-formula id="inf17">
<mml:math id="m26">
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is the lower limit of section <italic>s</italic>. The denominator number in the equation is a parameter that restricts the severity of punishment, and 10 is a suitable figure for restricting the power flow.<list list-type="simple">
<list-item>
<p>3) Load balance reward (positive reward) with the value range of <italic>A</italic>
<sub>2</sub>&#x2a;[0, 100]:</p>
</list-item>
</list>
<disp-formula id="e10">
<mml:math id="m27">
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>100</mml:mn>
<mml:mo>&#x2a;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>0.1</mml:mn>
<mml:mo>&#x2a;</mml:mo>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(10)</label>
</disp-formula>where <italic>A</italic>
<sub>2</sub> &#x3d; 3 is the score weight, <italic>L</italic> represents the total real load of the system at time <italic>t</italic>, and the denominator 0.1&#x2a;<italic>L</italic> is a normalization parameter that is set according to the comprehensive consideration of ultra-short-term forecast deviation and score interval. <italic>P</italic>
<sub>
<italic>i</italic>
</sub> is the active power output of generator unit <italic>i</italic>, and <italic>N</italic> is the total number of generator units.<list list-type="simple">
<list-item>
<p>4) Clean energy consumption reward (positive reward) with the value range of <italic>A</italic>
<sub>3</sub>&#x2a;[0, 100]:</p>
</list-item>
</list>
<disp-formula id="e11">
<mml:math id="m28">
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>&#x2a;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:mi mathvariant="italic">min</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(11)</label>
</disp-formula>where <italic>A</italic>
<sub>3</sub> is the score weight, <italic>P</italic>
<sub>
<italic>i</italic>
</sub> represents the active power output of the clean energy generator unit <italic>i</italic>, and <inline-formula id="inf18">
<mml:math id="m29">
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> represents the maximum output of the clean energy generator unit <italic>i</italic>. In order to avoid the denominator being 0 when calculating the score, when <inline-formula id="inf19">
<mml:math id="m30">
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is zero, the reward of generator unit <italic>i</italic> will be zero. There are a total of <italic>M</italic> clean energy generators.<list list-type="simple">
<list-item>
<p>5) Generator unit limit reward (positive reward) with the value range of <italic>A</italic>
<sub>4</sub>&#x2a;[0, 100]:</p>
</list-item>
</list>
<disp-formula id="e12">
<mml:math id="m31">
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:mi mathvariant="italic">max</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>100</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(12)</label>
</disp-formula>where <italic>A</italic>
<sub>4</sub> &#x3d; 1 is the score weight, <italic>N</italic> is the total number of units, and <italic>r</italic>
<sup>
<italic>i</italic>
</sup> is the reward value of the <italic>s</italic>th generator. It is calculated according to different situations (over-limit or normal). In over-limit situations (that is, exceeding the upper or lower limit of the predetermined value), severe penalties are imposed, whereas under normal circumstances, there is no penalty. The specific calculation method is as follows:<disp-formula id="e13">
<mml:math id="m32">
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="array">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(13)</label>
</disp-formula>where <italic>P</italic>
<sub>
<italic>i</italic>
</sub> is the active output of generator <italic>i</italic>, <inline-formula id="inf20">
<mml:math id="m33">
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is the upper limit of the active output of generator unit <italic>i</italic>, and <inline-formula id="inf21">
<mml:math id="m34">
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is the lower limit of the active output of generator <italic>i</italic>.</p>
</sec>
<sec id="s3-3">
<title>3.3 Constrained reinforcement learning loss</title>
<p>In the AC algorithm, the critic is trained to fit the reward. Its loss function is as follows:<disp-formula id="e14">
<mml:math id="m35">
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(14)</label>
</disp-formula>where <italic>G</italic>
<sub>
<italic>t</italic>
</sub> is <italic>R</italic>
<sub>
<italic>t</italic>&#x2b;1</sub> &#x2b; <italic>&#x3b3;R</italic>
<sub>
<italic>t</italic>&#x2b;2</sub> &#x2b; &#x22ef; &#x2b; <italic>&#x3b3;</italic>
<sup>
<italic>n</italic>&#x2212;1</sup>
<italic>R</italic>
<sub>
<italic>t</italic>&#x2b;<italic>n</italic>
</sub> &#x2b; <italic>&#x3b3;</italic>
<sup>
<italic>n</italic>
</sup>
<italic>Q</italic> (<italic>s</italic>
<sub>
<italic>t</italic>&#x2b;<italic>n</italic>
</sub>), and the actor is trained to find the optimal action <italic>a</italic> for the following minimization problem:<disp-formula id="e15">
<mml:math id="m36">
<mml:mtable class="align" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mtext>&#x2009;minimize&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>actor&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mspace width="0.3333em" class="nbsp"/>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mspace width="1em"/>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
<label>(15)</label>
</disp-formula>where <italic>w</italic>
<sub>1</sub>, <italic>w</italic>
<sub>2</sub>, <italic>w</italic>
<sub>3</sub>, and <italic>w</italic>
<sub>4</sub> are the weight values of each item, <italic>L</italic> represents the total load of the grid system, <italic>a</italic>
<sub>
<italic>i</italic>
</sub> represents the active power output of the <italic>i</italic>th generator, <italic>I</italic> represents the set of all generators, <italic>S</italic> represents the set of all grid sections, <italic>P</italic>
<sub>
<italic>s</italic>
</sub>(<italic>a</italic>) represents the power flow value of the section <italic>s</italic>, and <italic>a</italic> is the output of the generator unit. <inline-formula id="inf22">
<mml:math id="m37">
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is the maximum power flow of the section <italic>s</italic>, <italic>I</italic>
<sub>
<italic>ne</italic>
</sub> represents the clean energy generator set, <inline-formula id="inf23">
<mml:math id="m38">
<mml:msubsup>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> represents the maximum active output of the <italic>i</italic>th clean energy generator, and <italic>c</italic>
<sub>
<italic>i</italic>
</sub> represents the cost coefficient of the <italic>i</italic>th generator.</p>
<p>However, in the aforementioned formula, the two strong constraints, namely, load balance and power flow constraints, are regarded as objective functions with weights, which lead to the inability of the algorithm to obtain a stable solution in principle. Therefore, this paper proposes a constrained reinforcement learning loss (CRLL) algorithm as follows:<disp-formula id="e16">
<mml:math id="m39">
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
<mml:mspace width="0.3333em" class="nbsp"/>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mspace width="0.3333em" class="nbsp"/>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="equ1">
<mml:math id="m40">
<mml:mtext>&#x2009;s.t.&#x2009;</mml:mtext>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3c;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
</p>
<p>While satisfying the load balance and power flow constraints, the aforementioned objective functions can fit those actions that maximize clean energy consumption and minimize cost. It restores the essence of the grid section combination optimization problem, which is more conducive to the convergence of the reinforcement learning algorithm. This paper incorporates this loss into the training of the reinforcement learning algorithm by using Lagrangian constraints.</p>
</sec>
<sec id="s3-4">
<title>3.4 Training method and process</title>
<p>This paper chooses the actor&#x2013;critic reinforcement learning algorithm. The implementation of DRL combined with CRLL is shown in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>. The training process is as follows:<list list-type="simple">
<list-item>
<p>1) First, generate the sample data using the PYPOWER simulator. Then, clear the cache in the experience pool, set the initial state of the power system, and reset the reward value.</p>
</list-item>
<list-item>
<p>2) Input the observed state <italic>s</italic>
<sub>
<italic>t</italic>
</sub> of the current grid section system into the policy network, and obtain the active power output <italic>a</italic>
<sub>
<italic>t</italic>
</sub> of the generator unit through the policy network.</p>
</list-item>
<list-item>
<p>3) Input the output <italic>a</italic>
<sub>
<italic>t</italic>
</sub> of the generator unit into the reinforcement learning environment, and obtain the grid state <italic>s</italic>
<sub>
<italic>t</italic>&#x2b;1</sub> in the following stage, the reward value <italic>r</italic> corresponding to the current policy, and the completion state <italic>done</italic>.</p>
</list-item>
<list-item>
<p>4) Save the grid state <italic>s</italic>
<sub>
<italic>t</italic>
</sub>, the next moment&#x2019;s state <italic>s</italic>
<sub>
<italic>t</italic>&#x2b;1</sub>, the output policy <italic>a</italic>
<sub>
<italic>t</italic>
</sub>, the current reward value <italic>r</italic>, and the completion state <italic>done</italic> into the experience pool.</p>
</list-item>
<list-item>
<p>5) Judge whether the current experience pool has reached the upper limit of capacity. If the experience pool has not reached the limit, repeat Step 3; otherwise, go to Step 6.</p>
</list-item>
<list-item>
<p>6) When the accumulated data in the experience pool reach the batch size, they will be input into the policy network and value network as training data to train the network parameters. Then, return to Step 1.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s3-5">
<title>3.5 Annealing optimization algorithm</title>
<p>In the DRL method described previously, the action space is discretized, so the granularity of the action output by the model is not fine enough, resulting in the obtained solution being far beyond the optimal solution. In view of this problem, the annealing algorithm (<xref ref-type="bibr" rid="B4">Bakirtzis et al., 2002</xref>) is used after the proposed DRL algorithm to optimize the output of the generator unit. In this paper, we called it an annealing optimization algorithm. It can further improve the proposed DRL method to find the optimal fine-grained solution.</p>
<p>The annealing algorithm is a global optimization method based on a simulated physical annealing process. The basic idea of the algorithm is to start from an initial solution, continuously perturb the current solution randomly, and choose to accept the new solution or keep the current solution according to a certain probability. The function that accepts a new solution with a certain probability is called the &#x201c;acceptance criterion.&#x201d; The acceptance criterion allows the algorithm to perform a random walk in the search space and gradually reduces the temperature (that is, reduces the probability of accepting a new solution) until it reaches a stable state.</p>
<p>In the annealing optimization algorithm, the temperature parameter is usually used to control the variation in the acceptance criterion. At the beginning of the algorithm, the temperature is relatively high, so it tends to accept the new solutions according to the acceptance criterion. Therefore, a large-scale random search can be performed in the search space. As time goes by, the temperature gradually decreases, and it becomes much more difficult to accept the new solution, making the search process gradually stabilized. Eventually, the algorithm arrives at a near-optimal solution.</p>
<p>The annealing optimization algorithm is often used to solve nonlinear optimization problems, especially those with a large number of local optima. The advantage of the algorithm is that it can avoid falling into a local optimal solution and can perform a global search in the search space. In this paper, the annealing optimization algorithm is initialized by the output of the DRL model. The process of the annealing algorithm is as follows:<list list-type="simple">
<list-item>
<p>(1) Initialize the temperature <italic>T</italic> and the initial solution <italic>x</italic>.</p>
</list-item>
<list-item>
<p>(2) At the current temperature, produce a new solution <italic>x</italic>&#x2032; by using a random perturbation to the current solution.</p>
</list-item>
<list-item>
<p>(3) Calculate the energy difference &#x394;<italic>E</italic> between the new solution and the current solution.</p>
</list-item>
<list-item>
<p>(4) If &#x394;<italic>E</italic> &#x3c; 0, accept the new solution as the current solution.</p>
</list-item>
<list-item>
<p>(5) If &#x394;<italic>E</italic> &#x2265; 0, accept the new solution as the current solution with a probability <italic>P</italic> &#x3d; exp (&#x2212;&#x394;<italic>E</italic>/<italic>T</italic>).</p>
</list-item>
<list-item>
<p>(6) Lower the temperature <italic>T</italic>.</p>
</list-item>
<list-item>
<p>(7) Repeat Steps 2&#x2013;6 until the temperature drops to the end temperature or the maximum number of iterations is reached.</p>
</list-item>
</list>
</p>
<p>The algorithm flow chart is described in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Flow chart of the annealing optimization algorithm.</p>
</caption>
<graphic xlink:href="fenrg-11-1269854-g002.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4 Case study</title>
<p>To verify the effectiveness of the proposed method, this paper uses an IEEE 118-bus system. It consists of 118 buses, 54 generators, and 186 branches, representing a real power system network. The generators Gen 1 &#x223c;Gen 20 are set as the new energy units in this paper.</p>
<p>The computing environment is based on PYPOWER. The scheduling cycle is set to 15 min a day. According to the aforementioned description of MDP, the AC algorithm has 20 dimensions of state space. The dimension of the action space is set to 54. The detailed setting of hyper-parameters is shown in <xref ref-type="table" rid="T2">Table 2</xref>. The experiment runs on the Apple M1 Pro silicon with 8-core CPU and 16 GB memory. <xref ref-type="fig" rid="F3">Figure 3</xref> shows that the proposed model converges after approximately 900 episodes. As for the training time, the models converge after approximately 1 hour.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Setting of hyper-parameters.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Hyper-parameter</th>
<th align="center">Value</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Discount factor <italic>&#x3b4;</italic>
</td>
<td align="left">0.95</td>
</tr>
<tr>
<td align="left">Batch size</td>
<td align="left">64</td>
</tr>
<tr>
<td align="left">
<inline-formula id="inf31">
<mml:math id="m48">
<mml:mi>A</mml:mi>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mspace width="0.3333em" class="nbsp"/>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
<mml:mi>L</mml:mi>
<mml:mi>R</mml:mi>
</mml:math>
</inline-formula>
</td>
<td align="left">0.0001</td>
</tr>
<tr>
<td align="left">
<inline-formula id="inf32">
<mml:math id="m49">
<mml:mi>C</mml:mi>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mspace width="0.3333em" class="nbsp"/>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
<mml:mi>L</mml:mi>
<mml:mi>R</mml:mi>
</mml:math>
</inline-formula>
</td>
<td align="left">0.001</td>
</tr>
<tr>
<td align="left">
<italic>w</italic>
<sub>1</sub> in <inline-formula id="inf33">
<mml:math id="m50">
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mspace width="0.3333em" class="nbsp"/>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>
</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">
<italic>w</italic>
<sub>2</sub> in <inline-formula id="inf34">
<mml:math id="m51">
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mspace width="0.3333em" class="nbsp"/>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>
</td>
<td align="left">1</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Proposed model convergence after approximately 900 episodes.</p>
</caption>
<graphic xlink:href="fenrg-11-1269854-g003.tif"/>
</fig>
<p>Using the environment and the AC algorithm based on the CRLL in this paper, the agent maximizes the reward by adjusting the active power generated by the generator unit while minimizing the total cost and enhancing the new energy consumption. It can be seen from <xref ref-type="fig" rid="F4">Figure 4</xref> that the AC algorithm based on CRLL can converge and obtain the solution after 30 episodes. In contrast, it can be found that the traditional AC algorithm (vanilla AC) cannot achieve convergence within the same episode, and it cannot always reach the optimal solution. <xref ref-type="table" rid="T3">Table 3</xref> shows that vanilla AC training takes much longer than 4 h, but after using the proposed CRLL, the model convergence time is reduced to 1 h. By comparison, it can be seen that the proposed loss function plays a vital role in the stability of the solution and the convergence speed of model training.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Scheduling results of the AC algorithm based on CRLL and the traditional AC algorithm.</p>
</caption>
<graphic xlink:href="fenrg-11-1269854-g004.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Control experiment to verify the effectiveness of the proposed method. CRLL, constrained reinforcement learning loss; AO, annealing optimization algorithm.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center">Model convergence time</th>
<th align="center">System cost</th>
<th align="center">Power flow limit</th>
<th align="center">Load balance</th>
<th align="center">Clean energy consumption</th>
<th align="center">Generator unit limit</th>
<th align="center">Total reward score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Vanilla AC</td>
<td align="left">&#x3e;&#x3e;4 h</td>
<td align="left">50</td>
<td align="left">100</td>
<td align="left">150</td>
<td align="left">30</td>
<td align="left">50</td>
<td align="left">380</td>
</tr>
<tr>
<td align="left">Vanilla AC &#x2b; CRLL</td>
<td align="left">1 h</td>
<td align="left">55</td>
<td align="left">395</td>
<td align="left">295</td>
<td align="left">50</td>
<td align="left">95</td>
<td align="left">890</td>
</tr>
<tr>
<td align="left">Vanilla AC &#x2b; CRLL &#x2b; AO</td>
<td align="left">
<bold>1</bold> <bold>h</bold>
</td>
<td align="left">
<bold>63</bold>
</td>
<td align="left">
<bold>397</bold>
</td>
<td align="left">
<bold>287</bold>
</td>
<td align="left">
<bold>55</bold>
</td>
<td align="left">
<bold>98</bold>
</td>
<td align="left">
<bold>910</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values represent the best results in the experiment.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In <xref ref-type="table" rid="T3">Table 3</xref>, the experiment compares the results of three methods: 1) vanilla AC, 2) vanilla AC plus CRLL, and 3) vanilla AC plus CRLL and the annealing optimization algorithm. The scoring for all three methods is made up of five items. The full scores of system cost, power flow limit, load balance, clean energy consumption, and generator unit limit are 100, 400, 300, 100, and 100, respectively. Among them, power flow limit, load balance, and generator unit limit are strong constraints in the power grid section system. The goal of the proposed method is to make these three items close to full scores.</p>
<p>As shown in <xref ref-type="table" rid="T3">Table 3</xref>, the scores of all indicators have been greatly improved due to the proposed loss function, meeting the safety requirements of the power grid. The total reward score of the vanilla AC is 380, while the AC algorithm with CRLL achieves a higher total reward score of 890.</p>
<p>In addition, combining DRL with the annealing optimization algorithm further improved the accuracy of the solution. In <xref ref-type="table" rid="T3">Table 3</xref>, the average reward score of the final model is 910, among which the power flow limit, generator unit limit, and load balance rewards all reached almost full scores. It indicates that the addition of the annealing optimization algorithm further improves the performance of the algorithm and obtains a fine-grained optimal solution.</p>
<p>The results of the three methods are also shown in <xref ref-type="fig" rid="F5">Figure 5</xref> as a histogram. It intuitively demonstrates that the algorithm proposed in this paper is able to optimize the objective function under multiple strong constraints. So, it can be concluded that the method proposed in this paper is effective and can meet the requirements of online optimization and real-time control of the grid section.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Control experiment to verify the effectiveness of the proposed method.</p>
</caption>
<graphic xlink:href="fenrg-11-1269854-g005.tif"/>
</fig>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In the face of the high proportion of new energy generator units and complex constrained environments, this paper uses the deep reinforcement learning algorithm of simplified action space, together with CRLL, to search for the optimal active power output of generators. It also uses the annealing optimization algorithm to avoid the local optimal solution. The formulation and implementation process are introduced in detail. The test results on the IEEE 118-bus system show that the proposed method has good performance and is suitable for scheduling problems. In this paper, system cost and clean energy consumption have not reached full scores yet, and future improvement work can be committed to achieve better results. Another attempt is to use a multi-checkpoint and multi-process model inference approach that can both speed up the inference and improve the indicators by allowing each checkpoint to focus on different metrics.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>: AC training based on CRLL<list list-type="simple">
<list-item>
<p>
<bold>Require:</bold> episode <italic>ep</italic>, discount factor <italic>&#x3b3;</italic>, <italic>LR</italic>
<sub>
<italic>a</italic>
</sub>, <italic>LR</italic>
<sub>
<italic>c</italic>
</sub>, batch size <italic>b</italic>, <italic>&#x3b8;</italic>
<sub>
<italic>a</italic>
</sub>, <italic>&#x3b8;</italic>
<sub>
<italic>c</italic>
</sub>, <italic>maxsize</italic>
</p>
</list-item>
<list-item>
<p>&#x2003;1: &#x2003;<bold>while</bold> <italic>i</italic> &#x3c; <italic>ep</italic> <bold>do</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;2: &#x2003;&#x2003;reward &#x3d; 0; reset env; reset the experience pool</p>
</list-item>
<list-item>
<p>&#x2003;3: &#x2003;&#x2003;collect the trajectory information including (<italic>S</italic>
<sub>
<italic>t</italic>
</sub>, <italic>A</italic>
<sub>
<italic>t</italic>
</sub>, <italic>R</italic>
<sub>
<italic>t</italic>
</sub>, <italic>S</italic>
<sub>
<italic>t</italic>&#x2b;1</sub>)</p>
</list-item>
<list-item>
<p>&#x2003;4: &#x2003;&#x2003;<bold>if</bold> <italic>poolsize</italic> &#x3c; <italic>maxsize</italic> <bold>then</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;5: &#x2003;&#x2003;&#x2003;pool &#x2190; (<italic>S</italic>
<sub>
<italic>t</italic>
</sub>, <italic>A</italic>
<sub>
<italic>t</italic>
</sub>, <italic>R</italic>
<sub>
<italic>t</italic>
</sub>, <italic>S</italic>
<sub>
<italic>t</italic>&#x2b;1</sub>)</p>
</list-item>
<list-item>
<p>&#x2003;6: &#x2003;&#x2003;<bold>end if</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;7: &#x2003;&#x2003;<bold>if</bold> <italic>poolsize</italic> &#x3e; <italic>b</italic> <bold>then</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;8: &#x2003;&#x2003;&#x2003;update <italic>&#x3b8;</italic>
<sub>
<italic>a</italic>
</sub> with <italic>L</italic>
<sub>actor_constrained</sub>
</p>
</list-item>
<list-item>
<p>&#x2003;9: &#x2003;&#x2003;&#x2003;update <italic>&#x3b8;</italic>
<sub>
<italic>c</italic>
</sub> with <italic>L</italic>
<sub>critic</sub>
</p>
</list-item>
<list-item>
<p>&#x2003;10: &#x2003;&#x2003;<bold>end if</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;11: &#x2003;<bold>end while</bold>
</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. These data can be found at: <ext-link ext-link-type="uri" xlink:href="http://labs.ece.uw.edu/pstca/pf118/pg_tca118bus.htm">http://labs.ece.uw.edu/pstca/pf118/pg_tca118bus.htm</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>HZ: methodology, writing&#x2013;original draft, and conceptualization. ZW: conceptualization, software, and writing&#x2013;original draft. YH: writing&#x2013;original draft and investigation. QF: writing&#x2013;original draft and data curation. SL: writing&#x2013;original draft and methodology. GM: validation, writing&#x2013;original draft, and software. WL: formal analysis, visualization, and writing&#x2013;review and editing. QY: conceptualization, funding acquisition, resources, supervision, and writing&#x2013;review and editing.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alshammari</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Ramli</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Mehedi</surname>
<given-names>I. M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Hybrid chaotic maps-based artificial bee colony for solving wind energy-integrated power dispatch problem</article-title>. <source>Energies</source> <volume>15</volume>, <fpage>4578</fpage>. <pub-id pub-id-type="doi">10.3390/en15134578</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ardakani</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Bouffard</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Prediction of umbrella constraints</article-title>,&#x201d; in <conf-name>Proceedings of the 2018 Power Systems Computation Conference (PSCC)</conf-name>, <conf-loc>Dublin, Ireland</conf-loc>, <conf-date>June 2018</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>7</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ayd&#x131;n</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>&#xd6;zy&#xf6;n</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Solution to non-convex economic dispatch problem with valve point effects by incremental artificial bee colony with local search</article-title>. <source>Appl. Soft Comput.</source> <volume>13</volume>, <fpage>2456</fpage>&#x2013;<lpage>2466</lpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2012.12.002</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bakirtzis</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Biskas</surname>
<given-names>P. N.</given-names>
</name>
<name>
<surname>Zoumas</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Petridis</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Optimal power flow by enhanced genetic algorithm</article-title>. <source>IEEE Trans. power Syst.</source> <volume>17</volume>, <fpage>229</fpage>&#x2013;<lpage>236</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2002.1007886</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Diehl</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Warm-starting ac optimal power flow with graph neural networks</article-title>,&#x201d; in <conf-name>Proceedings of the 33rd Conference on Neural Information Processing Systems (NeurIPS 2019)</conf-name>, <conf-loc>Munich, Germany</conf-loc>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Gherbi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Lakdja</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Environmentally constrained economic dispatch via quadratic programming</article-title>,&#x201d; in <conf-name>Proceedings of the 2011 International Conference on Communications, Computing and Control Applications (CCCA)</conf-name>, <conf-loc>Hammamet, Tunisia</conf-loc>, <conf-date>March 2011</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Real-time decision making for power system via imitation learning and reinforcement learning</article-title>,&#x201d; in <conf-name>Proceedings of the 2022 IEEE/IAS Industrial and Commercial Power System Asia (I&#x26;CPS Asia)</conf-name>, <conf-loc>Shanghai, China</conf-loc>, <conf-date>July 2022</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>744</fpage>&#x2013;<lpage>748</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Irisarri</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kimball</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Clements</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Bagchi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Davis</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Economic dispatch with network and ramping constraints via interior point methods</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>13</volume>, <fpage>236</fpage>&#x2013;<lpage>242</lpage>. <pub-id pub-id-type="doi">10.1109/59.651641</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Intelligent optimization of reactive voltage for power grid with new energy based on deep reinforcement learning</article-title>,&#x201d; in <conf-name>Proceedings of the 2021 IEEE 5th Conference on Energy Internet and Energy System Integration (EI2)</conf-name>, <conf-loc>Taiyuan, China</conf-loc>, <conf-date>October 2021</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>2883</fpage>&#x2013;<lpage>2889</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Larouci</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ayad</surname>
<given-names>A. N. E. I.</given-names>
</name>
<name>
<surname>Alharbi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Alharbi</surname>
<given-names>T. E.</given-names>
</name>
<name>
<surname>Boudjella</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Tayeb</surname>
<given-names>A. S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Investigation on new metaheuristic algorithms for solving dynamic combined economic environmental dispatch problems</article-title>. <source>Sustainability</source> <volume>14</volume>, <fpage>5554</fpage>. <pub-id pub-id-type="doi">10.3390/su14095554</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Lillicrap</surname>
<given-names>T. P.</given-names>
</name>
<name>
<surname>Hunt</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Heess</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Erez</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tassa</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Continuous control with deep reinforcement learning</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1509.02971">https://arxiv.org/abs/1509.02971</ext-link>.</comment>
</citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>A deep reinforcement learning framework for automatic operation control of power system considering extreme weather events</article-title>,&#x201d; in <conf-name>Proceedings of the 2022 IEEE Power &#x26; Energy Society General Meeting (PESGM)</conf-name>, <conf-loc>Denver, CO, USA</conf-loc>, <conf-date>July 2022</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Mnih</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kavukcuoglu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Silver</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Graves</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Antonoglou</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Wierstra</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Playing atari with deep reinforcement learning</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1312.5602">https://arxiv.org/abs/1312.5602</ext-link>.</comment>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Modiri-Delshad</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kaboli</surname>
<given-names>S. H. A.</given-names>
</name>
<name>
<surname>Taslimi-Renani</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Abd Rahim</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Backtracking search algorithm for solving economic dispatch problems with valve-point effects and multiple fuel options</article-title>. <source>Energy</source> <volume>116</volume>, <fpage>637</fpage>&#x2013;<lpage>649</lpage>. <pub-id pub-id-type="doi">10.1016/j.energy.2016.09.140</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sayed</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Anis</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bi</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Feasibility constrained online calculation for real-time optimal power flow: A convex constrained deep reinforcement learning approach</article-title>. <source>IEEE Trans. Power Syst.</source>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2022.3220799</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shchetinin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>De Rubira</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>Hug</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>On the construction of linear approximations of line flow constraints for ac optimal power flow</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>34</volume>, <fpage>1182</fpage>&#x2013;<lpage>1192</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2018.2874173</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Silver</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lever</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Heess</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Degris</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wierstra</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Riedmiller</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Deterministic policy gradient algorithms</article-title>,&#x201d; in <conf-name>Proceedings of the International conference on machine learning (Pmlr)</conf-name>, <conf-loc>Beijing, China</conf-loc>, <conf-date>June 2014</conf-date>, <fpage>387</fpage>&#x2013;<lpage>395</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sutton</surname>
<given-names>R. S.</given-names>
</name>
</person-group> (<year>1988</year>). <article-title>Learning to predict by the methods of temporal differences</article-title>. <source>Mach. Learn.</source> <volume>3</volume>, <fpage>9</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1007/bf00115009</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sutton</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>McAllester</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mansour</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Policy gradient methods for reinforcement learning with function approximation</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>12</volume>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Real-time optimal power flow: A Lagrangian based deep reinforcement learning approach</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>35</volume>, <fpage>3270</fpage>&#x2013;<lpage>3273</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2020.2987292</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Relaxed deep learning for real-time economic generation dispatch and control with unified time scale</article-title>. <source>Energy</source> <volume>149</volume>, <fpage>11</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1016/j.energy.2018.01.165</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Fast <italic>&#x3bb;</italic>-iteration method for economic dispatch with prohibited operating zones</article-title>. <source>IEEE Trans. power Syst.</source> <volume>29</volume>, <fpage>990</fpage>&#x2013;<lpage>991</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2013.2287995</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>A graph-based deep reinforcement learning framework for autonomous power dispatch on power systems with changing topologies</article-title>,&#x201d; in <conf-name>Proceedings of the 2022 IEEE Sustainable Power and Energy Conference (iSPEC)</conf-name>, <conf-loc>Perth, Australia</conf-loc>, <conf-date>December 2022</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>W.-J.</given-names>
</name>
<name>
<surname>Diao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep reinforcement learning based real-time ac optimal power flow considering uncertainties</article-title>. <source>J. Mod. Power Syst. Clean Energy</source> <volume>10</volume>, <fpage>1098</fpage>&#x2013;<lpage>1109</lpage>. <pub-id pub-id-type="doi">10.35833/mpce.2020.000885</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zivic Djurovic</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Milacic</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Krsulja</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>A simplified model of quadratic cost function for thermal generators</article-title>,&#x201d; in <conf-name>Proceedings of the 23rd International DAAAM Symposium</conf-name>, <conf-loc>Zadar, Croatia</conf-loc>, <conf-date>October 2012</conf-date>, <fpage>24</fpage>&#x2013;<lpage>27</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>