<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Energy Res.</journal-id>
<journal-title>Frontiers in Energy Research</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Energy Res.</abbrev-journal-title>
<issn pub-type="epub">2296-598X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1377465</article-id>
<article-id pub-id-type="doi">10.3389/fenrg.2024.1377465</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Energy Research</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>RETRACTED: Generating adversarial deep reinforcement learning -based frequency control of Island City microgrid considering generalization of scenarios</article-title>
<alt-title alt-title-type="left-running-head">Wang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenrg.2024.1377465">10.3389/fenrg.2024.1377465</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wang</surname>
<given-names>Houtianfu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2641886/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Zhecong</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Qixin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>University of California, San Diego</institution>, <addr-line>San Diego</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>University of California, Los Angeles</institution>, <addr-line>Los Angeles</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1994319/overview">Yunqi Wang</ext-link>, Monash University, Australia</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1312399/overview">Linfei Yin</ext-link>, Guangxi University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2602892/overview">Cheng Yang</ext-link>, Shanghai University of Electric Power, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2642312/overview">Puliang Du</ext-link>, Southeast University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Houtianfu Wang, <email>houtianfuwang@sina.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>13</day>
<month>03</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="eretracted">
<day>12</day>
<month>11</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1377465</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>02</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Wang, Zhang and Wang.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Wang, Zhang and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The increasing incorporation of new energy sources into power grids introduces significant variability, complicating traditional load frequency control (LFC) methods. This variability can cause frequent load disturbances and severe frequency fluctuations in island city microgrids, leading to increased generation costs. To tackle these challenges, this paper introduces a novel Data knowledge-driven load frequency control (DKD-LFC) method, aimed at optimizing the balance between generation cost and frequency stability in isolated microgrids with high renewable energy integration. The DKD-LFC replaces conventional controllers with agent-based systems, utilizing reinforcement learning for adaptive frequency control in complex environments. A new policy generation algorithm, based on generative adversarial-proximal policy optimization (DAC-PPO), is proposed. This algorithm extends the traditional Actor-Critic framework of the Proximal Policy Optimization (PPO) by incorporating a Discriminator network. This network evaluates whether the input state-action pairs align with current or expert policies, guiding policy updates toward expert policies during training. Such an approach enhances the algorithm&#x2019;s generalization capability, crucial for effective LFC application in diverse operational contexts. The efficacy of the DKD-LFC method is validated using the isolated island city microgrid LFC model of the China Southern Grid (CSG), demonstrating its potential in managing the complexities of modern power grids.</p>
</abstract>
<kwd-group>
<kwd>load frequency control</kwd>
<kwd>generating adversarial deep reinforcement learning</kwd>
<kwd>isolated Island City microgrid</kwd>
<kwd>proximal policy optimization</kwd>
<kwd>discriminator network</kwd>
</kwd-group>
<counts>
<page-count count="11"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Sustainable Energy Systems</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>As a consequence of technological advancements, the global share of wind and photovoltaic (PV) power generation has significantly expanded. Owing to meteorological and temporal factors, these wind and PV plants are often regional and distributed in nature. The integration of microgrids, comprising distributed micro-sources and harnessing clean, renewable energies like wind and solar, is a pivotal trend shaping the future of electric power systems (<xref ref-type="bibr" rid="B1">Arya and Rai, 2022</xref>). Within these microgrids, distributed micro power sources&#x2014;including photovoltaic cells, wind turbines, and gas turbines primarily serve to provide electrical power (<xref ref-type="bibr" rid="B4">Gulzar et al., 2023</xref>). Converters, encompassing frequency converters, rectifiers, and inverters, play a crucial role in altering the form of electricity (<xref ref-type="bibr" rid="B5">Huang and Lv, 2023</xref>). The control system regulates various aspects of the microgrid, such as micro-sources, output voltage, power, energy storage, and loads, aiming to maintain a balance in voltage, power, and frequency within the microgrid. Energy storage devices within the microgrid are instrumental in managing the power equilibrium. Loads in the microgrid act by absorbing electrical energy and transforming it into other energy forms (<xref ref-type="bibr" rid="B18">Su et al., 2021</xref>). Typically, microgrids are interconnected with the larger grid at a common coupling point. This interconnection facilitates a flexible and reliable transition between islanded and grid-connected operational modes. It also helps in mitigating the impacts that may arise from the integration of numerous micro power sources into the grid. This progressive shift towards distributed, renewable energy sources and microgrids represents a transformative step in the evolution of modern power systems.</p>
<p>The ongoing expansion of modern microgrid infrastructures has led to a notable rise in new energy generation, intensifying the frequency regulation challenges in islanded microgrid operation modes. The reduction of frequency fluctuations is pivotal for ensuring the safe and stable functioning of microgrids. In islanded mode, microgrid operations can be disrupted by control inputs across various channels, rendering traditional control methods less effective for load frequency control. Contributing factors to load frequency variations include the randomness of local loads and the intermittency and uncertainty associated with new energy generation. Particularly in instances of shock loads within a microgrid, the increased reliance on new energy sources challenges the response capabilities and reserve capacity of conventional units, thereby complicating frequency regulation requirements.</p>
<p>In such contexts, maintaining load frequency stability post the integration of new energy sources becomes critically important. Load frequency control (LFC) in islanded microgrids primarily focuses on generating power regulation commands based on frequency deviations to maintain frequency within optimal ranges, which is crucial for the safe and stable operation of these systems. Various LFC methods have been proposed by researchers, including proportional-integral control (<xref ref-type="bibr" rid="B14">Patel et al., 2020</xref>), model predictive control (<xref ref-type="bibr" rid="B9">Li et al., 2022</xref>), adaptive control (<xref ref-type="bibr" rid="B12">Naderipour et al., 2019</xref>), sliding mode control (<xref ref-type="bibr" rid="B10">Li et al., 2023a</xref>), fuzzy control (<xref ref-type="bibr" rid="B3">Deshmukh et al., 2020</xref>), and robust control (<xref ref-type="bibr" rid="B8">Li et al., 2023b</xref>). Nonetheless, given the highly nonlinear and rapid-response nature of islanded microgrids, these methods often struggle to achieve multi-objective optimal coordinated control in complex stochastic environments. This is particularly challenging in scenarios with a significant presence of renewable power sources, where the intermittent and unpredictable output from these sources can significantly impact the frequency control performance and efficiency of the LFC system. Therefore, the development of advanced LFC strategies that can effectively handle the complexities introduced by renewable energy integration remains a critical area of research in the field of microgrid management.</p>
<p>Recent advancements in artificial intelligence (AI), particularly in power systems, have spotlighted the application of data-driven algorithms. Reinforcement Learning (RL), a notable AI paradigm, excels in decision-making in uncertain environments by learning from reward feedback for performance optimization (<xref ref-type="bibr" rid="B11">Mahmud et al., 2018</xref>). Deep Learning (DL) leverages multi-layer neural networks for effective data perception and representation through non-linear mapping (<xref ref-type="bibr" rid="B2">Cao et al., 2021</xref>). Combining these, Deep Reinforcement Learning (DRL) leverages both methodologies&#x2019; strengths, effectively addressing high-dimensional, time-varying, and nonlinear challenges in system optimization (<xref ref-type="bibr" rid="B13">Nguyen et al., 2020</xref>).</p>
<p>
<xref ref-type="bibr" rid="B6">Ismayil et al. (2015)</xref> investigated cutting-edge control strategies that integrate classic proportional-integral-derivative (PID) controllers with sophisticated optimization methods like genetic algorithms to enhance LFC performance. In a similar vein, <xref ref-type="bibr" rid="B17">Sharma et al. (2020)</xref> introduced the use of Artificial Neural Networks in LFC systems to improve control over nonlinear dynamics through the training of neuron connections using back-propagation gradient descent techniques. Furthermore, machine learning approaches, especially reinforcement learning, have been identified as highly effective in LFC, allowing systems to adaptively learn from trial-and-error, thus significantly enhancing control precision and efficiency.</p>
<p>
<xref ref-type="bibr" rid="B21">Yinsha et al. (2019)</xref> introduced Markov Decision Process (MDP)-based reinforcement learning models for single-task, multi-decision scenarios, incorporating negative feedback for collaborative strategy and task achievement. <xref ref-type="bibr" rid="B16">Sause (2013)</xref> demonstrated the efficacy of Q-learning and SARSA algorithms within a collaborative reinforcement framework for enhancing exploration in multi-intelligence resource competition. <xref ref-type="bibr" rid="B19">Ye et al. (2020)</xref> combined deep learning with deep deterministic policy gradients and prioritized empirical playback for excellence in complex state-action spaces. <xref ref-type="bibr" rid="B20">Yin et al. (2018)</xref> improved Q-learning&#x2019;s accuracy and stability with Double Q-Learning (DQL) algorithms, addressing the positive deviation issue critical for LFC system control.</p>
<p>In LFC, <xref ref-type="bibr" rid="B15">Peer et al. (2021)</xref> introduced Ensemble Bootstrapping for Q-Learning (EBSL), which mitigates variance and Q-value deviation during iterations, enhancing control precision. <xref ref-type="bibr" rid="B23">Yu et al. (2012)</xref> explored imitation learning strategies for islanded power systems in LFC, integrating eligibility traces with reinforcement learning for quicker convergence and better performance in complex environments. <xref ref-type="bibr" rid="B22">Yu et al. (2015)</xref> discussed multi-agent reinforcement learning for addressing interconnection and coordination issues, improving algorithmic efficiency. <xref ref-type="bibr" rid="B7">Khalid et al. (2022)</xref> utilized Improved Twin Delayed Deep Deterministic policy gradient (TD3) agents to refine PID controller parameters in multi-area interconnected systems, boosting stability and performance.</p>
<p>Amidst ongoing advancements, the issue of generalizability poses a significant obstacle in the realm of isolated microgrid Load Frequency Control (LFC). The essence of generalizability lies in the capacity of control systems or algorithms to adjust to a broad spectrum of conditions, particularly those beyond the scope of initial training scenarios. This attribute is indispensable within islanded microgrids, characterized by their fluctuating operational conditions and demand patterns. It is imperative for control systems to not only excel in familiar circumstances but also to adeptly navigate unanticipated events. The reliance on algorithms derived from historical data may prove inadequate in novel situations, highlighting the imperative for a synthesis of varied methodologies and the integration of reinforcement learning to elevate stability and adaptability amidst environmental shifts.</p>
<p>The current research introduces an innovative policy generation algorithm known as the Discriminator-Aided Actor-Critic Proximal Policy Optimization (DAC-PPO), which refines the conventional deep reinforcement learning paradigms. By integrating a Discriminator network into the established Proximal Policy Optimization (PPO) architecture, the algorithm distinguishes itself through its ability to evaluate if a given state-action pair is congruent with prevailing or expert policies, thereby steering the policy adaptation process towards expert-level proficiency throughout the training phase. This adjustment markedly augments the algorithm&#x2019;s aptitude for generalization, particularly within the context of Load Frequency Control (LFC) endeavors.</p>
<p>Furthering the advancements introduced by this sophisticated algorithm, the manuscript delineates the Data knowledge-driven Load Frequency Control (DKD-LFC) strategy. Aimed at achieving a balance between generation expenses and frequency stability in isolated microgrids, notably those with significant integration of renewable energy sources, DKD-LFC supplants conventional control mechanisms with agent-based systems that utilize adaptive reinforcement learning methodologies. The practical application of DKD-LFC within the isolated microgrid LFC framework of the China Southern Grid (CSG) is examined, illustrating its efficacy in orchestrating frequency control within intricate, renewable-dense microgrid configurations.</p>
<p>The main contributions of this paper are summarized as follows.<list list-type="simple">
<list-item>
<p>1) This research presents an innovative approach, termed the Data Knowledge-Driven Load Frequency Control (DKD-LFC) method, meticulously crafted to cater to the distinct needs of isolated microgrids, particularly those characterized by a significant incorporation of renewable energy sources. The essence of the DKD-LFC methodology lies in its strategic formulation aimed at achieving an optimal equilibrium between the operational costs associated with power generation and the imperative of maintaining frequency stability. This equilibrium is crucial in the context of microgrids heavily reliant on renewable energy sources, given their inherent variability and unpredictability. The DKD-LFC method addresses these challenges head-on, employing a sophisticated algorithm that dynamically adjusts to the fluctuating nature of renewable energy outputs, thereby ensuring a stable and efficient power supply while simultaneously managing to keep generation costs at a minimum. This dual focus not only enhances the operational efficiency of isolated microgrids but also contributes to the sustainable integration of renewable energy resources into the overall energy mix, marking a significant step forward in the pursuit of greener and more resilient power systems.</p>
</list-item>
<list-item>
<p>2) Furthermore, this study introduces a cutting-edge algorithmic development in the realm of policy generation, designated as the Discriminator-Aided Actor-Critic Proximal Policy Optimization (DAC-PPO). This refined version extends the foundational principles of the Proximal Policy Optimization (PPO), itself a cornerstone in the domain of conventional deep reinforcement learning paradigms. By embracing the Actor-Critic architecture inherent in the traditional PPO framework, the DAC-PPO method innovatively incorporates a Discriminator network into its operational schema. The primary function of this Discriminator is to rigorously evaluate whether a given input state-action pair is in congruence with either the prevailing policy or one derived from expert guidance. In effect, this Discriminator serves as a critical navigational beacon, steering the policy&#x2019;s developmental trajectory towards a level of expertise throughout the training process. Such a methodological advancement substantially bolsters the DAC-PPO algorithm&#x2019;s ability to adeptly generalize across a spectrum of LFC environments. This augmented capacity for generalization is pivotal, ensuring the algorithm&#x2019;s adaptability and successful deployment in a wide array of LFC scenarios, thereby marking a significant leap forward in the quest for more resilient and flexible control systems in the energy sector.</p>
</list-item>
</list>
</p>
<p>This paper is structured as follows: In <xref ref-type="sec" rid="s2">Section 2</xref>, we describe the model of the islanded microgrid; In <xref ref-type="sec" rid="s3">Section 3</xref>, we propose a novel method and explain its framework; In <xref ref-type="sec" rid="s4">Section 4</xref>, we perform case studies to assess the effectiveness of the method; and In <xref ref-type="sec" rid="s5">Section 5</xref>, we summarize the paper and discuss the main In <xref ref-type="sec" rid="s2">Section 2</xref>, we describe the model of the islanded microgrid; In <xref ref-type="sec" rid="s3">Section 3</xref>, we propose a novel method and explain its framework; In <xref ref-type="sec" rid="s4">Section 4</xref>, we perform case studies to assess the effectiveness of the method; and In <xref ref-type="sec" rid="s5">Section 5</xref>, we summarize the paper and discuss the main findings. In <xref ref-type="sec" rid="s5">Section 5</xref>, we summaries the paper and discuss the main findings.</p>
</sec>
<sec id="s2">
<title>2 Islanded microgrids and DKD-LFC model</title>
<sec id="s2-1">
<title>2.1 DKD-LFC model</title>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> illustrates a standalone microgrid model, encompassing diverse elements such as diesel engines, micro gas turbines, fuel cells, photovoltaic plants, wind turbines, energy storage systems, and loads, operating independently from the main grid. Frequency regulation in this microgrid is primarily managed by diesel engines and energy storage systems, while new energy units like wind and solar operate in maximum power tracking mode, contributing less to frequency regulation (<xref ref-type="bibr" rid="B20">Yin et al., 2018</xref>). The microgrid&#x2019;s controller distributes power among various sources to satisfy demand, ensuring economic, environmentally friendly, and stable operation.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>DKD-LFC model.</p>
</caption>
<graphic xlink:href="fenrg-12-1377465-g001.tif"/>
</fig>
<p>In this setup, diesel engines and energy storage units, offering more stable and controllable power, play a pivotal role in frequency regulation, especially during fluctuations. Conversely, wind turbines and photovoltaic arrays, subject to weather variability, provide less controllable power. The integration of these variable power sources increases the challenge of maintaining supply-demand balance and frequency stability within the microgrid.</p>
<p>Traditionally, centralized PI control has been employed for frequency regulation in microgrids, where frequency deviations are corrected by adjusting power distribution among generating units. However, the rise in new energy units has rendered traditional PI control insufficient for balancing power supply and load.</p>
<p>To address this, the paper proposes the Data knowledge-driven Load Frequency Control (DKD-LFC) method, designed to balance generation costs with frequency stability in isolated microgrids, especially those with high renewable energy integration. The DKD-LFC method replaces conventional controllers with agents that utilize reinforcement learning for adaptive frequency management in complex environments.</p>
</sec>
<sec id="s2-2">
<title>2.2 Unit modelling</title>
<sec id="s2-2-1">
<title>2.2.1 Diesel engine modelling</title>
<p>Diesel generators adjust their output through fuel supply regulation. This mechanism is pivotal in addressing frequency deviations, which signify an imbalance between load and supply. By modulating output, the diesel engine plays a crucial role in minimizing these frequency discrepancies. It is shown as Eq. (<xref ref-type="disp-formula" rid="e1">1</xref>).<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mtext>diesel</mml:mtext>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mtext>diesel</mml:mtext>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mtext>diesel</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the amount of power variation, <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mtext>diesel</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the scale factor and <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the frequency deviation.</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Micro gas turbines</title>
<p>Micro gas turbines modulate their power output by controlling gas flow, showcasing a rapid response to load variations, making them well-suited for frequency regulation tasks. Their power output adjustments are directly influenced by fluctuations in system frequency. These turbines exhibit unique dynamic response characteristics, distinguished by varying scaling and damping constants, differentiating them from diesel generators. It is shown as Eq. (<xref ref-type="disp-formula" rid="e2">2</xref>).<disp-formula id="e2">
<mml:math id="m5">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>gas</mml:mtext>
<mml:mo>&#x2006;</mml:mo>
<mml:mtext>turbine</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mtext>gas</mml:mtext>
<mml:mo>&#x2006;</mml:mo>
<mml:mtext>turbine</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mtext>gas</mml:mtext>
<mml:mo>&#x2006;</mml:mo>
<mml:mtext>turbine</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-2-3">
<title>2.2.3 Fuel cells</title>
<p>The output power of a fuel cell depends on the amount of fuel supplied, and the power can be adjusted by regulating the fuel flow. Fuel cells usually have good dynamic response characteristics. It is shown as Eq. (<xref ref-type="disp-formula" rid="e3">3</xref>).<disp-formula id="e3">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>fuel</mml:mtext>
<mml:mo>&#x2006;</mml:mo>
<mml:mtext>cell</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mtext>fuel</mml:mtext>
<mml:mo>&#x2006;</mml:mo>
<mml:mtext>cell</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mtext>fuel</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf4">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>fuel</mml:mtext>
<mml:mo>&#x2006;</mml:mo>
<mml:mtext>cell</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the power output of the fuel cell, <italic>F</italic>
<sub>
<italic>Fuel</italic>
</sub> is the fuel flow rate and <italic>K</italic>
<sub>
<italic>fuel cell</italic>
</sub> is the conversion efficiency.</p>
</sec>
<sec id="s2-2-4">
<title>2.2.4 Distributed PV/Wind aggregation modelling</title>
<p>Distributed photovoltaic usually works in maximum power point tracking mode, and its converter mostly adopts constant power control. The active dynamic transfer characteristic can be simplified to a first-order inertial link within the error tolerance as shown in Eq. (<xref ref-type="disp-formula" rid="e4">4</xref>).<disp-formula id="e4">
<mml:math id="m8">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>pv</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>s</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mtext>pv</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>solar</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf5">
<mml:math id="m9">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>pv</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the active output variation of the <italic>i</italic>th distributed PV converter, <inline-formula id="inf6">
<mml:math id="m10">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>solar</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is its active input variation, and <inline-formula id="inf7">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mtext>pv</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is its inertia time constant.</p>
<p>Since PV is a non-adjustable resource in this paper, the impact of its location distribution on control is not considered for the time being, and all PV units in the DVPP are considered as an equivalent PV plant for aggregation modelling.</p>
<p>Since the inertia time constant of the PV converter <inline-formula id="inf8">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mtext>pv</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a fast dynamic process compared to the control cycle, to simplify the analysis, it is considered that all distributed PV unit converters have the same time constant, i.e., <inline-formula id="inf9">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mtext>pv</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mtext>pv</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The active input/output model of the aggregated equivalent PV plant can be expressed as shown in Eq. (<xref ref-type="disp-formula" rid="e5">5</xref>).<disp-formula id="e5">
<mml:math id="m14">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mtext>PV</mml:mtext>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>s</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mtext>pv</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mtext>solar</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf10">
<mml:math id="m15">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mtext>PV</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the active output variation of all PV converters, <inline-formula id="inf11">
<mml:math id="m16">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mtext>solar</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the active input variation of all PVs.</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Generation costs</title>
<p>The calculation of generation cost is delineated as a comprehensive formula that quantifies the total expenses incurred in the production of electricity. This encompasses the aggregation of various operational costs associated with the generation process, including but not limited to, fuel expenses, maintenance of generation equipment, labor costs, and any additional overheads that directly contribute to the electricity production. The formula is meticulously designed to reflect the intricate dynamics of power generation, capturing both variable and fixed costs to provide a holistic overview of the financial implications of electricity production. By integrating these diverse cost factors, the formula offers a detailed insight into the economic considerations essential for efficient and sustainable power generation management. It is shown as Eq. (<xref ref-type="disp-formula" rid="e6">6</xref>).<disp-formula id="e6">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <italic>P</italic>
<sub>
<italic>Gi</italic>
</sub> is the output of the <italic>ith</italic> unit, <italic>a</italic>
<sub>
<italic>i</italic>
</sub>
<italic>, b</italic>
<sub>
<italic>i</italic>
</sub>
<italic>, c</italic>
<sub>
<italic>i</italic>
</sub> are constants, and <italic>C</italic>
<sub>
<italic>i</italic>
</sub> is the cost of the <italic>ith</italic> unit. It is shown as Eq. (<xref ref-type="disp-formula" rid="e7">7</xref>) and Eq. (<xref ref-type="disp-formula" rid="e8">8</xref>).<disp-formula id="e7">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>actual</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>plan</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m19">
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>plan</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>plan</mml:mtext>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>plan</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where &#x394;<italic>P</italic>
<sub>
<italic>Gi</italic>
</sub> is the regulation output of <italic>i</italic>th unit, <italic>P</italic>
<sub>
<italic>Gi</italic>
</sub>, <sub>actual</sub> is the output of <italic>i</italic>th unit, <italic>&#x3b1;</italic>
<sub>
<italic>i</italic>
</sub>, <italic>&#x3b2;</italic>
<sub>
<italic>i</italic>
</sub>, <italic>&#x3b3;</italic>
<sub>
<italic>i</italic>
</sub> are coefficients.</p>
</sec>
<sec id="s2-4">
<title>2.4 Objective functions and constraints</title>
<p>The DKD-LFC methodology is specifically designed to ensure the stability of grid frequency, which is paramount for the reliability and the overall quality of power within microgrid systems. The absence of precise frequency regulation can lead to significant adverse outcomes, including the risk of damage to critical infrastructure, a decline in the quality of the electricity provided, and the potential for widespread instability across the grid. Additionally, the costs associated with generating electricity have a profound influence on the operational dynamics of microgrid environments. Implementing a regime of efficient frequency control serves to reduce unnecessary energy consumption and lower operational costs, thereby improving the economic efficiency of the microgrid.</p>
<p>Islanded microgrids, characterized by their relatively modest scale and susceptibility to greater variability in load demands, pose unique challenges to maintaining consistent frequency control. These systems necessitate sophisticated management strategies that are capable of adjusting to the dual demands of minimizing operational costs while ensuring optimal performance. The DKD-LFC approach meets this requirement through the deployment of an integrated multi-objective optimization strategy. This strategy aims to balance the competing demands of cost-efficiency and performance by focusing on minimizing the combined impact of all relevant operational constraints.</p>
<p>By prioritizing both economic and performance-related considerations, the DKD-LFC approach delivers solutions that are both comprehensive in scope and highly adaptable to changing conditions. This balanced focus is essential for addressing the multifaceted challenges presented by islanded microgrids, ensuring that frequency stability is maintained without compromising on operational efficiency or economic viability. Through its implementation of multi-objective optimization, the DKD-LFC strategy effectively addresses these challenges, offering a nuanced approach that optimizes the balance between maintaining grid stability and managing generation costs. It is shown as Eq. (<xref ref-type="disp-formula" rid="e9">9</xref>) and Eq. (<xref ref-type="disp-formula" rid="e10">10</xref>).<disp-formula id="e9">
<mml:math id="m20">
<mml:mrow>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m21">
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>in</mml:mtext>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>in</mml:mtext>
</mml:msubsup>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>min</mml:mi>
</mml:msubsup>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>in</mml:mtext>
</mml:msubsup>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>max</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>rate</mml:mtext>
</mml:msubsup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>where &#x394;<italic>P</italic>
<sub>order-&#x2211;</sub> is the total command, &#x394;<italic>P</italic>
<sub>
<italic>i</italic>
</sub>
<sup>max</sup> and &#x394;<italic>P</italic>
<sub>
<italic>i</italic>
</sub>
<sup>min</sup> are the limits of the <italic>ith</italic> unit, &#x394;<italic>P</italic>
<sub>
<italic>i</italic>
</sub>
<sup>in</sup> is the command of the <italic>ith</italic> unit.</p>
</sec>
<sec id="s2-5">
<title>2.5 MDP modelling of DKD-LFCs</title>
<p>Deep Reinforcement Learning (DRL) synergizes deep neural networks with reinforcement learning, leveraging neural networks&#x2019; robust and rapid data representation and approximation capabilities for processing high-dimensional data. Concurrently, it employs reinforcement learning&#x2019;s decision-making faculties. The training of DRL models typically involves reinforcement learning algorithms, where decision-making is based on the current state and the corresponding value or policy function. These functions are iteratively updated through interaction with the environment and reception of reward signals, culminating in the accomplishment of the target task.</p>
<p>Within the reinforcement learning framework, an agent makes decisions based on the state of the external environment. The environment&#x2019;s attributes and its state possess Markov properties, indicating that a future state depends solely on the current state and is independent of past states. In other words, the response at a future time point (<italic>t</italic>&#x2b;1) is contingent only on the state and action at the present time (<italic>t</italic>). A reinforcement learning task that adheres to Markov properties is termed a Markov Decision Process (MDP). In an MDP, the decision-maker selects actions based on the current state, receives a reward, and transitions to the next state. MDP encompasses various elements:<list list-type="simple">
<list-item>
<p>- &#x201c;M&#x201d; represents the state dependency.</p>
</list-item>
<list-item>
<p>- &#x201c;D&#x201d; signifies the strategy determined by the agent, influencing state sequences through its actions and shaping future state developments in conjunction with environmental randomness,</p>
</list-item>
<list-item>
<p>- &#x201c;P&#x201d; denotes the time attribute, indicating that post-action, the environmental state changes, time advances, new states emerge, and this process perpetuates.</p>
</list-item>
</list>
</p>
<p>This framework of MDP forms the foundational structure for the decision-making process in reinforcement learning environments.</p>
<p>The MDP framework, central to reinforcement learning, comprises several key components:<list list-type="simple">
<list-item>
<p>1) State Space: This encompasses the entire set of potential states in which the agent can exist.</p>
</list-item>
<list-item>
<p>2) Action Space: This represents all possible actions accessible to the agent.</p>
</list-item>
<list-item>
<p>3) State Transition Probability: This is the likelihood of the agent transitioning to a subsequent state after executing action (<italic>a</italic>) in the current state (<italic>s</italic>).</p>
</list-item>
<list-item>
<p>4) Immediate Reward Function: This function quantifies the immediate reward the agent receives for taking action (a) in state (<italic>s</italic>).</p>
</list-item>
<list-item>
<p>5) Discount Factor: A factor indicating the extent to which future rewards are discounted, typically within the range of [(0,1)].</p>
</list-item>
</list>
</p>
<p>The primary objective of reinforcement learning in the context of a given MDP is to discover the optimal policy. For rapid load frequency control, it is essential to model the DKD-LFC using the DAC-PPO algorithm within the MDP framework.</p>
<sec id="s2-5-1">
<title>2.5.1 Action space</title>
<p>The sophisticated control system is required to simultaneously generate and dispatch precise regulatory directives to each unit within the designated areas, necessitating a complex action space for the controlling agent. This action space, detailed below, is structured to accommodate the intricate array of commands that ensure the seamless operation of each unit. It encapsulates the multifaceted decisions the agent must make, reflecting the high level of interactivity and coordination required for optimal system performance. It is shown as Eq. (<xref ref-type="disp-formula" rid="e11">11</xref>).<disp-formula id="e11">
<mml:math id="m22">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf12">
<mml:math id="m23">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the total command.</p>
</sec>
<sec id="s2-5-2">
<title>2.5.2 State space</title>
<p>The autonomous agent is tasked with closely monitoring the comprehensive dataset of the isolated microgrid&#x2019;s operational status, executing decisions to effectively manage any deviations in frequency based on the real-time and historical state observations. It diligently tracks the instantaneous generation dynamics of every unit&#x2019;s turbine. This vigilant surveillance is crucial, especially to address the challenging scenario where there is an absence of a rapid-response unit capable of adjusting to a succession of significant perturbations. To craft a strategic response to such intricate situations, the local state space is meticulously structured to encapsulate the following parameters. This structured approach ensures that the agent has access to a detailed and multi-faceted view of the microgrid&#x2019;s performance, empowering it to take corrective measures when faced with complex disturbance patterns, thus maintaining the integrity and stability of the power system. It is shown as Eq. (<xref ref-type="disp-formula" rid="e12">12</xref>).<disp-formula id="e12">
<mml:math id="m24">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mi>t</mml:mi>
</mml:msubsup>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2006;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>where <inline-formula id="inf13">
<mml:math id="m25">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the total power output.</p>
</sec>
<sec id="s2-5-3">
<title>2.5.3 Reward functions</title>
<p>Frequency deviation and generation cost are used as reward functions, and a penalty factor is added to accelerate the training since frequency tuning failures can occur during the exploration of the action. It is shown as Eq. (<xref ref-type="disp-formula" rid="e13">13</xref>) and Eq. (<xref ref-type="disp-formula" rid="e14">14</xref>).<disp-formula id="e13">
<mml:math id="m26">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m27">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0.05</mml:mn>
<mml:mi>H</mml:mi>
<mml:mi>Z</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0.05</mml:mn>
<mml:mi>H</mml:mi>
<mml:mi>Z</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>where <italic>r</italic> is the reward and <italic>A</italic> is the punishment function.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<title>3 y DAC-PPO algorithm based DKD-LFC application</title>
<sec id="s3-1">
<title>3.1 Optimisation algorithm for proximal strategies</title>
<p>Drawing on the advancements in reinforcement learning and imitation learning within the realms of flight control and intelligent gaming, this study introduces the DAC-PPO. This algorithm is specifically designed to tackle the challenges of low convergence efficiency and suboptimal utilization of expert experience, which are prevalent in conventional reinforcement learning algorithms for generating air combat maneuver strategies. The DAC-PPO algorithm enhances the standard Proximal Policy Optimization (PPO) by integrating a Discriminator network into the Actor-Critic framework. This Discriminator network is tasked with discerning whether the input state-action pair is derived from the current or an expert strategy.</p>
<p>Reinforcement learning algorithms include value-based, policy-based and combined Actor-Critic methods. This paper is based on the Actor-Critic method. Actor network is the strategy network, denoted as <inline-formula id="inf14">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf15">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the <inline-formula id="inf16">
<mml:math id="m30">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> moment state, <italic>&#x3b8;</italic> denotes the strategy network parameters, and the strategy network outputs the action Critic network is the value network, the reward is denoted as shown in Eq. (<xref ref-type="disp-formula" rid="e15">15</xref>).<disp-formula id="e15">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mi>t</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>&#x221e;</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:msup>
<mml:mi>t</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msup>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:msup>
<mml:mi>t</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:msup>
<mml:mi>t</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>where <italic>E</italic> (.) is the mathematical expectation, <italic>&#x3b3;</italic> is the discount factor, which ensures that the Markov decision process can converge; <italic>r</italic> is the reward function, which is usually designed based on the experience of the experts in the real environment. The goal of reinforcement learning algorithms is to maximize the return on rounds. Among many algorithms, TRPO (<xref ref-type="bibr" rid="B21">Yinsha et al., 2019</xref>), PPO (<xref ref-type="bibr" rid="B16">Sause, 2013</xref>) and other algorithms have high stability and high convergence efficiency, which have become typical baseline algorithms.</p>
<p>It adopts the dominance function <inline-formula id="inf17">
<mml:math id="m32">
<mml:mrow>
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to represent the strategy advantages and disadvantages, in order to reduce the variance and improve the stability of the algorithm. The definition is as shown in Eq. (<xref ref-type="disp-formula" rid="e16">16</xref>).<disp-formula id="e16">
<mml:math id="m33">
<mml:mrow>
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
</p>
<p>In practice, <inline-formula id="inf18">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is defined to estimate <inline-formula id="inf19">
<mml:math id="m35">
<mml:mrow>
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, using the widely used generalized advantage estimation (GAE) method, defined as shown in Eq. (<xref ref-type="disp-formula" rid="e17">17</xref>).<disp-formula id="e17">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>where <inline-formula id="inf20">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, the parameter <italic>&#x3bb;</italic> is used to balance the variance and bias.</p>
<p>In addition, the algorithm uses importance sampling to directly pre-crop the probability magnitude of the old strategy and the new strategy, denoted as <inline-formula id="inf21">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>ok</mml:mtext>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Therefore, the loss function of PPO algorithm is expressed as Eqs <xref ref-type="disp-formula" rid="e18">(18)</xref>&#x2013;<xref ref-type="disp-formula" rid="e20">(20)</xref>.<disp-formula id="e18">
<mml:math id="m39">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mtext>ppo</mml:mtext>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mtext>policy</mml:mtext>
<mml:mtext>ppo</mml:mtext>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mtext>value</mml:mtext>
<mml:mtext>ppo</mml:mtext>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>
<disp-formula id="e19">
<mml:math id="m40">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mtext>policy</mml:mtext>
<mml:mtext>ppo</mml:mtext>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>clip</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>
<disp-formula id="e20">
<mml:math id="m41">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mtext>value</mml:mtext>
<mml:mtext>ppo</mml:mtext>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>R</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>&#x3c6;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(20)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-2">
<title>3.2 Generating adversarial imitation learning algorithms</title>
<p>The Generative Adversarial Imitation Learning (GAIL) algorithm is inspired by Maximum Entropy Inverse Reinforcement Learning (IRL) and Generative Adversarial Networks (GAN). Based on the framework of on-policy algorithms (e.g., TRPO, PPO, etc.), the discriminator <italic>D</italic>
<sub>
<italic>&#x3c9;</italic>
</sub>(<italic>s</italic>
<sub>
<italic>t</italic>
</sub>
<italic>, a</italic>
<sub>
<italic>t</italic>
</sub>) is designed to determine whether the input sampled data is generated from expert strategy or current strategy. The goal of GAIL algorithm can be understood as matching the distribution of current strategy with the distribution of expert strategy, so that the discriminator can&#x2019;t distinguish between the current strategy and the expert strategy, and its loss function is defined as:<disp-formula id="e21">
<mml:math id="m42">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mtext>dise</mml:mtext>
<mml:mtext>pil</mml:mtext>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>ln</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b5;</mml:mi>
</mml:msub>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>ln</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(21)</label>
</disp-formula>
<disp-formula id="e22">
<mml:math id="m43">
<mml:mrow>
<mml:mfenced open="" close="]" separators="|">
<mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mtext>policy</mml:mtext>
<mml:mtext>gail</mml:mtext>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>ln</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:msup>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(22)</label>
</disp-formula>
</p>
<p>In the GAIL algorithm, firstly, the current policy <inline-formula id="inf22">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the expert policy <inline-formula id="inf23">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are sampled to update the discriminator parameter <inline-formula id="inf24">
<mml:math id="m46">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>; then, the policy network parameter <italic>&#x3b8;</italic> is updated to maximise the output of the discriminator, and <inline-formula id="inf25">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:msup>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is analogous to the state-action value function <italic>Q</italic> (<italic>s</italic>,<italic>a</italic>) in the reinforcement learning algorithm. The Generative Adversarial Imitation Learning (GAIL) algorithm&#x2019;s dependency on expert data for policy generation raises concerns about the performance of these policies, especially when the dataset includes sub-optimal policies or fails to meet objectives. Addressing this issue, this study proposes a Generative Adversarial Proximal Policy Optimization algorithm, which merges the exploratory strengths of reinforcement learning environments with the policy constraint benefits inherent in imitation learning.</p>
</sec>
<sec id="s3-3">
<title>3.3 DAC-PPO algorithm</title>
<p>In the DKD-LFC based on the DAC-PPO algorithm, the Q-function of the critic, which evaluates the quality of the actions, is modelled as shown in Eq. (<xref ref-type="disp-formula" rid="e23">23</xref>).<disp-formula id="e23">
<mml:math id="m48">
<mml:mrow>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3bc;</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(23)</label>
</disp-formula>
</p>
<p>The block diagram of the DAC-PPO algorithm is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. The model consists of a value network, a strategy network, and a discriminator network, and only the strategy network is retained when deploying the model; the experience pool consists of an example experience pool and a round experience pool, and the trajectory data ternary in the example pool <inline-formula id="inf26">
<mml:math id="m49">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="normal">E</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is generated by the human-machine confrontation and the machine-machine confrontation based on the rule model. The circular experience pool in this model captures trajectory quaternions produced through the interaction of the current strategy with the environment, and is reset after each training cycle. This model encompasses three distinct data flows:<list list-type="simple">
<list-item>
<p>1) Environment Interaction Data Flow: Here, the current strategy engages with the environment, generating trajectory data that is stored in the circular experience pool.</p>
</list-item>
<list-item>
<p>2) Discriminator and Strategy Network Update Data Flow: Post-training round, the parameters of the discriminator network are updated using the gradient descent method, as specified in Eq. <xref ref-type="disp-formula" rid="e21">21</xref>. Subsequently, Eq. <xref ref-type="disp-formula" rid="e22">22</xref> guides the update of the strategy network&#x2019;s parameters, steering the current strategy distribution towards convergence with the expert strategy.</p>
</list-item>
<list-item>
<p>3) Network Update Data Flow: This follows the Proximal Policy Optimization (PPO) algorithm&#x2019;s process. The PPO algorithm updates the Actor-Critic (AC) network in line with Eq. <xref ref-type="disp-formula" rid="e22">22</xref>, maintaining consistency with the established PPO framework.</p>
</list-item>
</list>
</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Farmwork of DAC-PPO algorithm.</p>
</caption>
<graphic xlink:href="fenrg-12-1377465-g002.tif"/>
</fig>
<p>In order to significantly improve the speed at which the algorithm converges, as well as its overall stability, the methodology incorporates a sophisticated distributed parallel computing framework. This innovative approach is characterized by the deployment of multiple distributed rollout workers, specifically &#x201c;n&#x201d; in number, alongside a singular central learning unit. These rollout workers are tasked with directly interacting with the designated environment, during which they meticulously gather data pertaining to trajectories over a specific round of operation. Once this data collection phase is concluded for a round, each rollout worker proceeds to calculate the gradient based on the strategies they have executed. This calculated gradient information is then efficiently relayed back to the central learning entity, where it undergoes a process of gradient aggregation.</p>
<p>Subsequent to the aggregation process, the central learner updates the network parameters to reflect the newly accumulated gradient information. These updated parameters are promptly disseminated back to all the distributed rollout workers. This ensures that each worker is equipped with the latest network adjustments, enabling them to initiate the process of gathering fresh data for the upcoming round. This cycle of data collection, gradient computation, aggregation, and parameter dissemination not only fosters a rapid convergence rate but also significantly bolsters the algorithm&#x2019;s stability by leveraging the parallel processing capabilities of the distributed computing setup. Through this methodical approach, the algorithm benefits from a heightened efficiency in learning and adaptation, showcasing the effectiveness of integrating distributed computing techniques for complex computational tasks.</p>
<p>The flow of the algorithm is shown below. Firstly, establish the example experience pool <inline-formula id="inf27">
<mml:math id="m50">
<mml:mrow>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf28">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the <italic>n</italic> flight trajectory, i.e., <inline-formula id="inf29">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>s</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>k</mml:mi>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>a</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>k</mml:mi>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mover accent="true">
<mml:mo>,</mml:mo>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Initialise the network parameters and hyperparameters of the algorithm. At the end of each round, sample <inline-formula id="inf30">
<mml:math id="m53">
<mml:mrow>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf31">
<mml:math id="m54">
<mml:mrow>
<mml:msubsup>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, calculate the policy gradients <inline-formula id="inf32">
<mml:math id="m55">
<mml:mrow>
<mml:mo>&#x2207;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>gail</mml:mtext>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf33">
<mml:math id="m56">
<mml:mrow>
<mml:mo>&#x2207;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>ppo</mml:mtext>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, and the learner accumulates the gradients and updates the network parameters, and finally, output the optimal policy network parameters <inline-formula id="inf34">
<mml:math id="m57">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Case studies</title>
<p>This study conducts comprehensive simulations to evaluate the effectiveness of the Data knowledge-driven Load Frequency Control (DKD-LFC) method, based on the Discriminator-Aided Actor-Critic Proximal Policy Optimization (DAC-PPO) algorithm. The research involves a detailed comparative analysis of DKD-LFC against various control algorithms including the PPO controller, TRPO controller, TD3 controller, Deep Deterministic Policy Gradient (DDPG) optimized controller, Particle Swarm Optimization (PSO) optimized fuzzy-PI controller, and Genetic Algorithm (GA) PI controller.</p>
<p>For these simulations, a robust control system is employed, featuring a high-capacity computer equipped with dual 2.10&#xa0;GHz Intel Xeon Platinum processors and 16&#xa0;GB of memory. The simulations are conducted using MATLAB/Simulink software, version 9.8.0 (R2020a), providing a solid platform for a meticulous evaluation of the proposed method. This testing framework facilitates a comprehensive examination of the DKD-LFC&#x2019;s performance, enabling a clear comparison with established control strategies in the domain.</p>
<p>The simulation designed to assess the DKD-LFC model for an isolated urban megacity microgrid confronts several complexities. These include incorporating wind turbines (WT), photovoltaic (PV) systems, and addressing the effects of irregular step load disturbances. Conducted over an extended period of 7,200&#xa0;s, the simulation offers ample opportunity to observe and evaluate the system&#x2019;s response to these diverse challenges. The results are presented in a detailed graph, showcasing the system&#x2019;s dynamic behavior and its capability to manage the intricacies posed by these various energy sources and load variations. This graphical representation is instrumental in gauging the efficiency and resilience of the microgrid&#x2019;s LFC system under such demanding conditions.</p>
<p>As detailed in <xref ref-type="table" rid="T1">Table 1</xref>, the DAC-PPO algorithm demonstrates a significant reduction in frequency deviation (13.3%&#x2013;99.3%) and generation cost (0.0012%&#x2013;0.098%). <xref ref-type="fig" rid="F3">Figure 3</xref> highlights the employment of a prioritised replay technique by DAC-PPO in its pre-learning phase, enhancing strategy robustness. This technique ensures rapid response and power shortage compensation. Ensures rapid response and power shortage compensation by each unit during disturbances. According to <xref ref-type="fig" rid="F4">Figure 4</xref>, this is due to the high generalisation of the DAC-PPO algorithm. The DAC-PPO algorithm, which has better performance in the face of different load disturbances, and thus does not lead to overshoot of the total regulated output. DAC-PPO outperforms other algorithms with lower mean frequency deviation and reduced output overshoot, indicating its superior robustness. Conversely, the DDPG algorithm, due to its simplistic empirical replay strategy, fails to achieve optimal LFC strategy. The PPO algorithm&#x2019;s DKD-LFC framework, while exhibiting less frequency deviation and reduced output overshoot of the total regulated output. Framework, while exhibiting less frequency deviation variation, suffers in control performance variability under different disturbances because it lacks robustness-enhancing techniques in its pre-learning phase. <xref ref-type="table" rid="T1">Table 1</xref> further reveals DAC-PPO&#x2019;s superiority in minimising total cost, attributed <xref ref-type="table" rid="T1">Table 1</xref> further reveals DAC-PPO&#x2019;s superiority in minimising total cost, attributed to its cost-reduction focus during the control process, ensuring more stable generation costs. The fuzzy-based algorithm, neglecting multi-objective optimality of frequency deviation and generation cost, and relying on basic fuzzy-based algorithms, has been shown to have a low resilience under various disturbances. Cost, and relying on basic fuzzy rules, leads to inconsistent frequency regulation performance, particularly noticeable in its significant overshoot During the second disturbance. Overall, DAC-PPO maintains consistent performance across random disturbances, showcasing the fastest frequency This efficiency positions DAC-PPO as the most effective in terms of lowest average frequency deviation relative to total generation cost. This efficiency positions DAC-PPO as the most effective in terms of lowest average frequency deviation relative to total generation cost.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Statistical results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Control algorithm</th>
<th align="center">Average frequency deviation (HZ)</th>
<th align="center">Power generation cost ($)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">DAC-PPO</td>
<td align="center">0.00996</td>
<td align="center">4,654.96</td>
</tr>
<tr>
<td align="left">PPO</td>
<td align="center">0.01107</td>
<td align="center">4,658.05</td>
</tr>
<tr>
<td align="left">TRPO</td>
<td align="center">0.01466</td>
<td align="center">4,657.78</td>
</tr>
<tr>
<td align="left">TD3</td>
<td align="center">0.01439</td>
<td align="center">4,657.28</td>
</tr>
<tr>
<td align="left">DDPG</td>
<td align="center">0.01609</td>
<td align="center">4,656.76</td>
</tr>
<tr>
<td align="left">PSO-Fuzzy-PI</td>
<td align="center">0.01512</td>
<td align="center">4,656.62</td>
</tr>
<tr>
<td align="left">GA-PI</td>
<td align="center">0.01642</td>
<td align="center">4,656.66</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th align="left">Control algorithm</th>
<th align="center">Average frequency deviation (HZ)</th>
<th align="center">Power generation cost ($)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">DAC-PPO</td>
<td align="center">0.01930</td>
<td align="center">8,210.97</td>
</tr>
<tr>
<td align="left">PPO</td>
<td align="center">0.02146</td>
<td align="center">8,216.62</td>
</tr>
<tr>
<td align="left">TRPO</td>
<td align="center">0.02753</td>
<td align="center">8,216.00</td>
</tr>
<tr>
<td align="left">TD3</td>
<td align="center">0.02863</td>
<td align="center">8,214.96</td>
</tr>
<tr>
<td align="left">DDPG</td>
<td align="center">0.03175</td>
<td align="center">8,213.87</td>
</tr>
<tr>
<td align="left">PSO-Fuzzy-PI</td>
<td align="center">0.03023</td>
<td align="center">8,213.50</td>
</tr>
<tr>
<td align="left">GA-PI</td>
<td align="center">0.03309</td>
<td align="center">8,213.59</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Frequency deviation.</p>
</caption>
<graphic xlink:href="fenrg-12-1377465-g003.tif"/>
</fig>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Total regulated output.</p>
</caption>
<graphic xlink:href="fenrg-12-1377465-g004.tif"/>
</fig>
<sec id="s4-1">
<title>4.1 Case 2: step disturbance and renewable disturbance</title>
<p>In this paper, a sophisticated smart distribution grid model is developed, incorporating a diverse array of new energy sources to assess the control performance of the DAC-PPO algorithm in an environment characterized by high stochasticity. The model integrates novel energy sources like electric vehicles, wind power, small hydropower, micro gas turbines, fuel cells, photovoltaics, and biomass. Given their unpredictability, electric vehicles, wind power, and photovoltaics can be used to control a diverse array of new energy sources in an environment characterised by high stochasticity, wind power, and photovoltaic power are modelled as random load disturbances, not impacting the system&#x2019;s frequency regulation. The wind power output from turbines is modelled using finite element method (FEM). From turbines is modelled using finite bandwidth white noise to replicate random wind patterns. Similarly, the active output of the photovoltaic power generation is modelled by simulating the frequency regulation of the system. Similarly, the active output of the photovoltaic power generation is modelled by simulating the daily variation in light intensity. Detailed parameters for each energy unit are provided in (<xref ref-type="bibr" rid="B10">Li et al., 2023a</xref>). This approach allows for a comprehensive analysis of the DAC-PPO&#x2019;s performance under varying and unpredictable energy inputs.</p>
<p>The provided table showcases the simulation statistics, highlighting the generation cost as the cumulative regulation cost of all generators within a day. In these simulations, the DAC-PPO algorithm exhibits superior performance compared to other algorithms. It achieves 1.11&#x2013;1.53 times lower It achieves 1.11&#x2013;1.53 times lower frequency deviation and a 0.0425%&#x2013;0.0629% reduction in generation cost, as indicated by the distribution network data. DAC-PPO also excels in aspects of economy, self-adaptation, and coordinated optimisation, surpassing other intelligent algorithms. Its robustness and efficacy are further validated through tests under various disturbances. Validated through tests under various disturbances, including step, square, and random waveforms. These tests reveal DAC-PPO&#x2019;s high convergence, learning efficiency, and adaptability, showcasing DAC-PPO&#x2019;s ability to adapt to the changing environment. Learning efficiency, and adaptability, showcasing its ability to withstand random disturbances and enhance dynamic control within the given environment.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>This work presents the following main contributions:<list list-type="simple">
<list-item>
<p>1) This paper presents a Data Knowledge-Driven Load Frequency Control (DKD-LFC) approach. DKD-LFC is designed to navigate the trade-off between generation cost and frequency stability in isolated microgrids with a high penetration of renewable energy sources.</p>
</list-item>
<list-item>
<p>2) This paper proposes a policy generation algorithm (DAC-PPO) based on Generative Adversarial Proximal Policy Optimisation (GAPPO) based on conventional deep reinforcement learning algorithms. Based on the Actor-Critic framework of the traditional PPO algorithm, a Discriminator network is added to determine whether the input state-action belongs to the current policy or the expert policy, and to constrain the current policy to be updated in the direction of the expert policy during policy training. This technique is used to improve the generalisation of the algorithm to ensure the high generalisation of LFC to the scenarios.</p>
</list-item>
</list>
</p>
<p>This study conducts a comprehensive evaluation of the DKD-LFC method and DAC-PPO algorithm within the island microgrid LFC model of the China South Grid, comparing them against various existing algorithms. This study conducts a comprehensive evaluation of the DKD-LFC method and DAC-PPO algorithm within the island microgrid LFC model of the China South Grid, comparing them against various existing algorithms. They demonstrate the quickest frequency response and minimal overshoot when subjected to a range of random disturbances. Moreover, they achieve the lowest average frequency deviation, particularly when considering the total generation cost, highlighting their efficiency and effectiveness in microgrid management.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s11">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>HW: Conceptualization, Data curation, Formal Analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing. ZZ: Conceptualization, Data curation, Formal Analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing. QW: Conceptualization, Data curation, Formal Analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fenrg.2024.1377465/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fenrg.2024.1377465/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.XLSX" id="SM1" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arya</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Rai</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Cascade FOPI-FOPTID controller with energy storage devices for AGC performance advancement of electric power systems</article-title>. <source>Sustain. Energy Technol. Assessments</source> <volume>53</volume>, <fpage>102671</fpage>. <pub-id pub-id-type="doi">10.1016/j.seta.2022.102671</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>C. T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Weak human preference supervision for deep reinforcement learning</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>32</volume> (<issue>12</issue>), <fpage>5369</fpage>&#x2013;<lpage>5378</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2021.3084198</pub-id>
<pub-id pub-id-type="pmid">34101604</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deshmukh</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Ballal</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Suryawanshi</surname>
<given-names>H. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A fuzzy logic based supervisory control for power management in multibus DC microgrid</article-title>. <source>IEEE Trans. Industry Appl.</source> <volume>56</volume> (<issue>6</issue>), <fpage>6174</fpage>&#x2013;<lpage>6185</lpage>. <pub-id pub-id-type="doi">10.1109/TIA.2020.3012415</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Gulzar</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Umar</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Al-Dhaifallah</surname>
<given-names>M. M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Robust load frequency control of hybrid power system</article-title>,&#x201d; in <conf-name>2023 International Conference on Control, Automation and Diagnosis (ICCAD)</conf-name>, <conf-loc>Rome, Italy</conf-loc>, <conf-date>May 2023</conf-date>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Load frequency control of power system based on improved AFSA-PSO event-triggering scheme</article-title>. <source>Front. Energy Res.</source> <volume>11</volume>. <pub-id pub-id-type="doi">10.3389/fenrg.2023.1235467</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ismayil</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Sindhu</surname>
<given-names>T. K.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Optimal fractional order PID controller for automatic generation control of two-area power systems</article-title>. <source>Int. Trans. Electr. Energ. Syst.</source> <volume>25</volume> (<issue>12</issue>), <fpage>3329</fpage>&#x2013;<lpage>3348</lpage>. <pub-id pub-id-type="doi">10.1002/etep.2038</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khalid</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ramli</surname>
<given-names>M. a. M.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Hidayat</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Efficient load frequency control of renewable integrated power system: a Twin delayed DDPG-based deep reinforcement learning approach</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>1051561</fpage>&#x2013;<lpage>1051574</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2022.3174625</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023b</year>). <article-title>Optimal dual-model controller of solid oxide fuel cell output voltage using imitation distributed deep reinforcement learning</article-title>. <source>Int. J. Hydrog. Energy</source> <volume>48</volume> (<issue>37</issue>), <fpage>14053</fpage>&#x2013;<lpage>14067</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijhydene.2022.12.194</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Coordinated load frequency control of multi-area integrated energy system using multi-agent deep reinforcement learning</article-title>. <source>Appl. Energy</source> <volume>306</volume>, <fpage>117900</fpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2021.117900</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>Brain-inspired deep meta-reinforcement learning for active coordinated fault-tolerant load frequency control of multi-area grids</article-title>. <source>IEEE Trans. Autom. Sci. Eng.</source> <volume>1</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1109/TASE.2023.3263005</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahmud</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kaiser</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Hussain</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vassanelli</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Applications of deep learning and reinforcement learning to biological data</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>29</volume> (<issue>6</issue>), <fpage>2063</fpage>&#x2013;<lpage>2079</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2018.2790388</pub-id>
<pub-id pub-id-type="pmid">29771663</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Naderipour</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abdul-Malek</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ramachandaramurthy</surname>
<given-names>V. K.</given-names>
</name>
<name>
<surname>Kalam</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Miveh</surname>
<given-names>M. R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Hierarchical control strategy for a three-phase 4-wire microgrid under unbalanced and nonlinear load conditions</article-title>. <source>ISA Trans.</source> <volume>94</volume>, <fpage>94352</fpage>&#x2013;<lpage>94369</lpage>. <pub-id pub-id-type="doi">10.1016/j.isatra.2019.04.025</pub-id>
<pub-id pub-id-type="pmid">31078293</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>N. D.</given-names>
</name>
<name>
<surname>Nahavandi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep reinforcement learning for multiagent systems: a review of challenges, solutions, and applications</article-title>. <source>IEEE Trans. Cybern.</source> <volume>50</volume> (<issue>9</issue>), <fpage>3826</fpage>&#x2013;<lpage>3839</lpage>. <pub-id pub-id-type="doi">10.1109/TCYB.2020.2977374</pub-id>
<pub-id pub-id-type="pmid">32203045</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patel</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Meegahapola</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Mcgrath</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Automatic generation control of multi-area power system with network constraints and communication delays</article-title>. <source>J. Mod. Power Syst. Clean. Energy</source> <volume>8</volume> (<issue>3</issue>), <fpage>454</fpage>&#x2013;<lpage>463</lpage>. <pub-id pub-id-type="doi">10.35833/MPCE.2018.000513</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Peer</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Tessler</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Merlis</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Meir</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Ensemble bootstrapping for Q-learning</article-title>,&#x201d; in <source>Proceedings of the 38th international conference on machine learning</source>. Editors <person-group person-group-type="editor">
<name>
<surname>MARINA,</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>TONG</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<publisher-name>PMLR</publisher-name>, <publisher-loc>Breckenridge, CO, USA</publisher-loc>), <fpage>8454</fpage>&#x2013;<lpage>8463</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sause</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Coordinated reinforcement learning agents in a multi-agent virtual environment</article-title>,&#x201d; in <conf-name>2013 12th International Conference on Machine Learning and Applications</conf-name>, <conf-loc>Miami, FL, USA</conf-loc>, <conf-date>December 2013</conf-date>, <fpage>227</fpage>&#x2013;<lpage>230</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharma</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Panwar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Arya</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kumawat</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Integrating layered recurrent ANN with robust control strategy for diverse operating conditions of AGC of the power system</article-title>. <source>IET Gener. Transm. Distrib.</source> <volume>14</volume> (<issue>18</issue>), <fpage>3886</fpage>&#x2013;<lpage>3895</lpage>. <pub-id pub-id-type="doi">10.1049/iet-gtd.2019.0935</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Optimization and H &#x221e; performance analysis for load frequency control of power systems with time-varying delays</article-title>. <source>Front. Energy Res.</source> <volume>9</volume>, <fpage>762480</fpage>. <pub-id pub-id-type="doi">10.3389/fenrg.2021.762480</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Papadaskalopoulos</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Strbac</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep reinforcement learning for strategic bidding in electricity markets</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>11</volume> (<issue>2</issue>), <fpage>1343</fpage>&#x2013;<lpage>1355</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2019.2936142</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Design of a novel smart generation controller based on deep Q learning for large-scale interconnected power system</article-title>. <source>J. Energy Chem.</source> <volume>144</volume> (<issue>3</issue>), 04018033. <pub-id pub-id-type="doi">10.1061/(ASCE)EY.1943-7897.0000519</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yinsha</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wenyi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhiwen</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Research on PSO-fuzzy algorithm optimized control for multi-area AGC system with DFIG wind turbine</article-title>,&#x201d; in <conf-name>2019 14th IEEE Conference on Industrial Electronics and Applications (ICIEA)</conf-name>, <conf-loc>Xi&#x27;an, China</conf-loc>, <conf-date>June 2019</conf-date>, <fpage>877</fpage>&#x2013;<lpage>881</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H. Z.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Multi-agent correlated equilibrium Q(&#x3bb;) learning for coordinated smart generation control of interconnected power grids</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>30</volume> (<issue>4</issue>), <fpage>1669</fpage>&#x2013;<lpage>1679</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2014.2357079</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q. H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>R(&#x3bb;) imitation learning for automatic generation control of interconnected power grids</article-title>. <source>Automatica</source> <volume>48</volume> (<issue>9</issue>), <fpage>2130</fpage>&#x2013;<lpage>2136</lpage>. <pub-id pub-id-type="doi">10.1016/j.automatica.2012.05.043</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>