<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Energy Res.</journal-id>
<journal-title>Frontiers in Energy Research</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Energy Res.</abbrev-journal-title>
<issn pub-type="epub">2296-598X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1384995</article-id>
<article-id pub-id-type="doi">10.3389/fenrg.2024.1384995</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Energy Research</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Optimizing load frequency control in isolated island city microgrids: a deep graph reinforcement learning approach with data enhancement across extensive scenarios</article-title>
<alt-title alt-title-type="left-running-head">Wu et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenrg.2024.1384995">10.3389/fenrg.2024.1384995</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Min</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Ma</surname>
<given-names>Dakui</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2654150/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xiong</surname>
<given-names>Kaiqing</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yuan</surname>
<given-names>Linkun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Dongfang Electronics Corporation</institution>, <addr-line>Yantai</addr-line>, <addr-line>Shandong</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Guangzhou Power Supply Bureau of Guangdong Power Grid Co., Ltd.</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1994319/overview">Yunqi Wang</ext-link>, Monash University, Australia</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2602892/overview">Cheng Yang</ext-link>, Shanghai University of Electric Power, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2642312/overview">Puliang Du</ext-link>, Shanghai University of Electric Power, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1312399/overview">Linfei Yin</ext-link>, Guangxi University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2165798/overview">Tao Zhou</ext-link>, Nanjing University of Science and Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Dakui Ma, <email>madakui_csg@gdcsg.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>02</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1384995</elocation-id>
<history>
<date date-type="received">
<day>11</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Wu, Ma, Xiong and Yuan.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wu, Ma, Xiong and Yuan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>This study presents a Data-Enhanced Optimum Load Frequency Control (DEO-LFC) strategy for microgrids, targeting an optimal balance between generation costs and frequency stability amidst high renewable energy integration. By replacing traditional controls with agent-based systems and reinforcement learning, the DEO-LFC employs an optimal balance between generation costs and frequency stability amidst high renewable energy integration. By replacing traditional controls with agent-based systems and reinforcement learning, the DEO-LFC employs a Soft Graph Actor Critic (SGAC) algorithm, integrating deep reinforcement learning with graph sequence neural networks for effective frequency management. Proven effective in the China Southern Grid&#x2019;s island microgrid model, DEO-LFC offers a sophisticated solution to the challenges posed by the island microgrid model. Proven effective in the China Southern Grid&#x2019;s island microgrid model, DEO-LFC offers a sophisticated solution to the challenges posed by the variability of modern power grids.</p>
</abstract>
<kwd-group>
<kwd>load frequency control</kwd>
<kwd>deep graph reinforcement learning</kwd>
<kwd>isolated island city microgrid</kwd>
<kwd>soft graph actor critic</kwd>
<kwd>data-enhanced</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Sustainable Energy Systems</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>In the context of escalating concerns over fossil fuel depletion, the importance of renewable energy in enhancing smart grid capabilities has surged. Renewable energy sources, inherently constrained by environmental conditions and geographical dispersion, necessitate integration into the power grid via sophisticated inverter technologies, leading to the development of Distributed Generation (DG) (<xref ref-type="bibr" rid="B15">Wang et al., 2013</xref>). This shift towards distributed generation presents a stark contrast to traditional centralized power generation systems, offering notable benefits in terms of energy efficiency, environmental sustainability, operational flexibility, reliability, and economic viability.</p>
<p>However, the integration of renewable energy into the power grid introduces challenges related to its unpredictable output, characterized by random intermittency and volatility (<xref ref-type="bibr" rid="B9">Mahboob Ul Hassan et al., 2022</xref>). Such unpredictability can compromise the power quality and jeopardize the stability of the grid system (<xref ref-type="bibr" rid="B14">Su et al., 2021</xref>). To address these issues and harness the full potential of distributed generation, the microgrid concept has been proposed. As an advanced technological solution predicated on renewable distributed power generation, microgrids are poised to play a pivotal role in the evolution of smart grid infrastructures. They facilitate the integration of diverse small-scale distributed energy resources and loads, ensuring safe and reliable operation both in grid-connected and islanded modes (<xref ref-type="bibr" rid="B5">Huang and Lv, 2023</xref>).</p>
<p>In grid-connected mode, microgrids complement the utility grid by supplying power to local loads and potentially exporting surplus energy back to the grid. Conversely, in scenarios of grid failure or disturbances, microgrids transition to islanded mode, independently powering local loads. This operational flexibility, however, necessitates robust control strategies to maintain system stability in the absence of grid support, given that islanded microgrids (IMGs) rely heavily on renewable energy sources (RESs) connected via power electronic converters. This configuration results in a diminished system inertia, posing challenges for frequency stability, reliable power supply, and efficient renewable energy utilization (<xref ref-type="bibr" rid="B4">Hosseini and Etemadi, 2008</xref>).</p>
<p>Load Frequency Control (LFC) emerges as a critical mechanism within power systems to balance frequency and active power demand across specific control areas (<xref ref-type="bibr" rid="B1">Bengiamin and Chan, 1982</xref>). Achieving optimal LFC performance in islanded microgrids requires a nuanced approach that not only improves frequency control but also minimizes the generation costs associated with distributed energy resources. Traditional LFC strategies, such as proportional-integral control (<xref ref-type="bibr" rid="B11">Mi et al., 2013</xref>), model predictive control (<xref ref-type="bibr" rid="B10">Mi et al., 2016</xref>), and adaptive control (<xref ref-type="bibr" rid="B2">Chen et al., 1991</xref>), often struggle to meet these dual objectives effectively.</p>
<p>Therefore, this discourse underscores the imperative for innovative control strategies that can adeptly manage the unique challenges posed by the integration of renewable energy sources into microgrids. The advancement of microgrid technology and the optimization of LFC mechanisms are essential for realizing the full potential of renewable energy within the smart grid paradigm, ensuring both environmental sustainability and grid stability.</p>
<sec id="s1-1">
<title>1.1 Proportional-integral control</title>
<p>Initial LFC studies stem from the Proportional-Integral (PI) control era, valued for their simplicity and computational ease (Wang et al., 1993). Integrated into LFC, PI controls split into steady-state integral and transient proportional parts. The PI control&#x2019;s widespread use in LFC hinges on its non-differential regulation efficacy in fundamental power setups. However, as power systems grow and face more stochastic disruptions, PI&#x2019;s static nature limits its dynamic stability, threatening frequency equilibrium and risking failures (<xref ref-type="bibr" rid="B17">Wang et al., 1994</xref>). Scholars have since sought to improve PI for LFC, with (<xref ref-type="bibr" rid="B3">Chen et al., 2022</xref>) incorporating sliding mode control for disturbance resilience. The rise of complex, nonlinear, and interconnected power systems demands control strategies that address these traits. <xref ref-type="bibr" rid="B8">Long et al. (2021)</xref> introduces a tri-layer LFC model for detailed power system dynamics, with a control strategy for nonlinear management. Yet, its reliance on specific parameters and a model-centric approach limits broad use. This shift from PI to adaptive, interconnected strategies reflects the ongoing effort to manage modern power systems&#x2019; complexities. The ongoing evolution of LFC methods highlights the necessity for flexible, robust, and efficient controls to maintain the stability and reliability of our increasingly intricate and interconnected power grids.</p>
</sec>
<sec id="s1-2">
<title>1.2 Model predictive control and adaptive control</title>
<p>Model Predictive Control (MPC) uses dynamic models, typically linear empirical ones, to predict and optimize system behavior over a future time span, adjusting the present state with future constraints in mind (<xref ref-type="bibr" rid="B12">Peng et al., 2023</xref>). This allows for real-time feedback and corrections. A distributed MPC algorithm promotes collaborative LFC between wind and thermal plants, improving overall system performance through dynamic cooperation.</p>
<p>Adaptive Control (AC), on the other hand, adjusts its parameters and rules in response to changing system conditions, maintaining stability despite uncertainties and significant disturbances without needing known variability bounds (<xref ref-type="bibr" rid="B23">Yan et al., 2022</xref>). AC in LFC, via adaptive dynamic programming, minimizes frequency deviations in grids, requiring less reliance on prior knowledge than MPC. However, AC systems tend to be more complex and costly.</p>
<p>Traditional controls often lack adaptability, affecting the performance and cost of frequency control in islanded microgrids. AI algorithms offer a solution, handling nonlinear complexities and improving operational efficiency and stability. AI learns and adapts to changing conditions, refining LFC strategies, enhancing energy efficiency, and ensuring reliable power supply. AI&#x2019;s predictive abilities also help prevent system failures, increasing microgrid resilience to uncertainties and disturbances.</p>
</sec>
<sec id="s1-3">
<title>1.3 Artificial intelligence control</title>
<p>In the modern era of islanded microgrids, which significantly integrate renewable energy resources, the complexity and interconnectedness of information flow across various regions necessitate a systematic approach for prioritizing the use of novel energy sources. Traditional LFC strategies face challenges in navigating the intricate decision-making processes required for efficient energy management. Within the realm of computer science, Artificial Intelligence (AI) emerges as a critical field, striving to emulate human cognitive abilities, including learning, decision-making, and problem-solving. The inherent capability of AI to engage with and learn from its environment independently positions it as a formidable tool for addressing complex challenges in energy systems.</p>
<p>The integration of AI with LFC mechanisms represents a pioneering effort to transcend the limitations of conventional LFC methods. For example, the work of <xref ref-type="bibr" rid="B6">Jia et al. (2019)</xref> showcases the successful application of Q-learning to LFC, significantly enhancing system adaptability through the ongoing refinement of the state-action matrix for comprehensive power control in simplified models. Furthermore, <xref ref-type="bibr" rid="B27">Yu et al. (2012)</xref> have introduced an innovative imitation learning strategy that integrates eligibility traces into reinforcement learning, yielding improved LFC performance in islanded systems through faster convergence and enhanced dynamic capabilities. In another notable advancement, <xref ref-type="bibr" rid="B25">Yu et al. (2015)</xref> have developed a cooperative reinforcement learning strategy, employing multiple intelligent agents to devise an optimal unified control strategy, effectively addressing the challenges posed by the interconnectivity of disparate control regions.</p>
<p>Additionally, <xref ref-type="bibr" rid="B28">Zhang et al. (2023a)</xref> have constructed a tri-level architecture for a multi-agent system, enabling coordinated control over LFC and Automatic Voltage Control (AVC). This architecture leverages the autonomous, independent, and collaborative nature of intelligent agents to ensure logical consistency while decentralizing control functions physically. In a groundbreaking approach, <xref ref-type="bibr" rid="B18">Xi et al. (2018)</xref> propose the Evolutionary Population Cooperative Control (EPCC) strategy, utilizing a win-lose criterion and space-time tunneling concept to quickly achieve Nash equilibrium within a multi-agent system (MIS) framework. This strategy, rooted in the Multi-Agent System Stochastic Consensus Game (MAS-SCG), promotes frequent information exchanges among intelligent agents, demonstrating the potential of AI to enhance decision-making and operational efficiency in complex energy systems.</p>
<p>Reinforcement Learning (RL) is a key machine learning paradigm that focuses on devising strategies for agents to make optimal decisions to meet set objectives through interactions with environmental states. Utilizing a Markov decision process, RL entails recognizing states and selecting actions guided by rewards, leading to state transitions. Studies have investigated applying power system&#x2019;s instantaneous frequency and transmission line power flow as RL environmental states, with power allocation directives as action decisions, tackling power allocation challenges effectively. (<xref ref-type="bibr" rid="B26">Yu et al., 2011</xref>; <xref ref-type="bibr" rid="B13">Shangguan et al., 2021</xref>; <xref ref-type="bibr" rid="B29">Zhang et al., 2023b</xref>).</p>
<p>
<xref ref-type="bibr" rid="B31">Zhang et al. (2020)</xref> have considered the total profit of power generation companies, incorporating dispatch mileage compensation into power command allocation to enhance the economic efficiency of power generation. Li et al. (<xref ref-type="bibr" rid="B30">Zhang et al., 2021</xref>) introduced an adaptive distributed auction algorithm for optimizing LFC power command allocation, minimizing the deviation between total and allocated power commands. This method is praised for its rapid convergence and model-free nature, ensuring precise generator power control. Moreover, <xref ref-type="bibr" rid="B7">Li et al. (2021)</xref> proposed a double-delay deep deterministic policy gradient algorithm, augmented by a multi-experience pool probabilistic replay strategy, improving controller training efficiency, action instruction quality, and mitigating stochastic perturbations in systems incorporating new energy sources, highlighting the evolving applications of RL in power system optimization.</p>
<p>In the dynamically evolving context of islanded microgrids, enriched with a diverse array of renewable energy resources, the exploration of AI control strategies combined with RL distribution tactics is underway to realize an intelligently integrated LFC system across multiple regions. This research endeavor has led to the development of multi-regional, multi-layered distributed LFC frameworks, enabling intelligence dissemination from macro to micro levels (<xref ref-type="bibr" rid="B20">Xi et al., 2016a</xref>; <xref ref-type="bibr" rid="B22">Xi et al., 2016b</xref>). To address these limitations and enhance generalizability, <xref ref-type="bibr" rid="B19">Xi et al. (2021)</xref> replaced the traditional wolf climbing LFC algorithm with PDWoLF-PHC, proposing a VWPS-HDC method that offers improved performance through time-consistent climbing. Additionally, to overcome the drawbacks of the WPH algorithm, <xref ref-type="bibr" rid="B21">Xi et al. (2022)</xref> introduced a cost-consistent VWPC-HDC method, achieving faster dynamic optimization, enhanced robustness, and reduced generation costs.</p>
<p>However, the practicality of these methodologies, based on the wolf pack hunting principle, is constrained by their reliance on extensive knowledge systems. To mitigate these limitations.</p>
<p>The challenge of ensuring wide-ranging applicability in the domain of standalone microgrid Load Frequency Control (LFC) remains a critical issue. It necessitates the creation of control frameworks and algorithms that can effectively operate in diverse scenarios beyond the scope of their original design. This adaptability is essential for managing the dynamic operational landscapes and the variability in demand that are characteristic of isolated microgrids. The integration of a diverse set of techniques, alongside reinforcement learning, is vital for enhancing the robustness and adaptability required to navigate changes in the environment.</p>
<p>This paper introduces the Data-Enhanced Optimum Load Frequency Control (DEO-LFC) methodology, which is designed to achieve a harmonious balance between generation costs and frequency stability in microgrids with a substantial integration of renewable energy sources. The Soft Graph Actor Critic (SGAC) algorithm is presented as a groundbreaking fusion of deep reinforcement learning and graph sequence neural network models, tailored to manage the intricacies of adaptive frequency regulation. By employing a Markov decision process for system modeling and a graph to sequence neural network for policy function approximation, the DEO-LFC approach highlights its potential impact. Its application to the isolated island city microgrid model within the China Southern Grid serves as a testament to its effectiveness in modern electrical grid settings.</p>
<p>The main contributions of this paper are summarized as follows.<list list-type="simple">
<list-item>
<p>1) Introduction to the DEO-LFC Methodology: The DEO-LFC methodology signifies a paramount advancement in the realm of frequency stability enhancement and cost minimization in isolated microgrids, especially those with substantial renewable energy sources integration. This methodological shift toward employing agent-based systems, which utilize reinforcement learning algorithms, marks a departure from conventional control strategies. The DEO-LFC framework presents an innovative, adaptive approach to managing frequency control challenges in complex operational contexts. It specifically targets the issues arising from the fluctuating nature of renewable energy sources, thereby facilitating a more reliable and cost-effective energy management system.</p>
</list-item>
<list-item>
<p>2) Creation of the SGAC Algorithm: Central to the DEO-LFC methodology is the groundbreaking creation of the SGAC (Simultaneous Graph-based Actor-Critic) algorithm. This state-of-the-art algorithm fuses the sophistication of deep reinforcement learning with the nuanced processing capabilities of graph sequence neural networks, making it uniquely equipped to navigate the complexities of load frequency control. The algorithm employs a Markov decision process for comprehensive system modeling and is further enhanced by the integration of advanced iterative learning techniques. The SGAC algorithm&#x2019;s design is purposefully crafted to devise an optimal strategy for frequency management, showcasing an innovative approach that elevates the performance and reliability of modern electrical power grids.</p>
</list-item>
</list>
</p>
<p>The organisation of this manuscript is as follows: <xref ref-type="sec" rid="s2">Section 2</xref> delineates the configuration of the islanded microgrid system. Subsequently, <xref ref-type="sec" rid="s3">Section 3</xref> introduces a groundbreaking approach, detailing its structural framework. <xref ref-type="sec" rid="s2">Section 2</xref> delineates the configuration of the islanded microgrid system. <xref ref-type="sec" rid="s4">Section 4</xref> is dedicated to the execution of case studies designed to evaluate the proposed method&#x2019;s efficacy. Finally, <xref ref-type="sec" rid="s5">Section 5</xref> concludes the document by providing a comprehensive summary and discussing the principal outcomes derived from the research conducted. Finally, <xref ref-type="sec" rid="s5">Section 5</xref> concludes the document by providing a comprehensive summary and discussing the principal outcomes derived from the research conducted.</p>
</sec>
</sec>
<sec id="s2">
<title>2 Islanded microgrids and DEO-LFC model</title>
<sec id="s2-1">
<title>2.1 DEO-LFC model</title>
<p>In microgrids, integration of Distributed Generation (DG) units such as Photovoltaic (PV), Wind Power (WP), and Energy Storage (ES) systems is achieved via grid-connected inverter interfaces, which allow these units to align with desired power outputs through specific control mechanisms. A simplified model for these inverters is used to explain the Load Frequency Control (LFC) framework, highlighting the role of traditional, renewable, and storage energy sources in frequency regulation.</p>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> illustrates an autonomous microgrid setup featuring diverse generation sources like diesel engines, micro gas turbines, fuel cells, PV, wind turbines, ES systems, and consumer loads. Here, diesel engines and ES systems play a crucial role in frequency regulation, while renewables focus on maximizing power output through Maximum Power Point Tracking (MPPT), offering limited frequency support. The control system of the microgrid dynamically distributes power to match demand, prioritizing efficiency, sustainability, and stability.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>DEO-LFC model.</p>
</caption>
<graphic xlink:href="fenrg-12-1384995-g001.tif"/>
</fig>
<p>For independent operation, microgrids require self-adjusting power generation for voltage and frequency stability, incorporating Primary Frequency Control (PFC) and LFC mechanisms. PFC deals with immediate power output adjustments in response to frequency changes, whereas LFC involves coordinated efforts across multiple sources to correct frequency discrepancies, typically managed by centralized controllers and communication systems. ES and diesel generators are key to microgrid frequency stability, with PV and WP units focusing on MPPT due to their variable output.</p>
<p>Recent research suggests strategies for integrating wind and solar into frequency regulation by reserving part of their output to improve system response. Yet, the focus remains on PFC. This study explores how diesel and ES significantly contribute to frequency stability, managing variances in power supply. Wind and solar, despite their fluctuating nature, are considered less reliable for maintaining balance and stability in microgrids.</p>
<p>This paper introduces a DEO-LFC method designed to optimize generation costs while ensuring frequency stability in microgrids rich in renewables. The DEO-LFC strategy balances cost-efficiency with the critical need for frequency stability, addressing the challenges posed by high renewable energy integration in isolated microgrids.</p>
<p>The framework employs advanced algorithms for adaptive frequency regulation, adept at navigating the complex dynamics characteristic of such systems. It promises improved performance, especially in mitigating the unpredictability associated with renewable energy sources. By integrating data-driven insights and knowledge-based control, the DEO-LFC approach enhances the reliability and efficiency of frequency management in microgrids, aligning operational expenditures with the overarching goal of frequency stabilization. This methodological innovation stands to significantly advance the operational robustness of isolated microgrids, ensuring stability amidst the fluctuating nature of renewable energy contributions.</p>
</sec>
<sec id="s2-2">
<title>2.2 Unit modelling</title>
<sec id="s2-2-1">
<title>2.2.1 Diesel engine modelling</title>
<p>Diesel generators (DGs) serve as pivotal controllable Distributed Generation (DG) units within microgrids, offering low operational costs and high reliability but posing environmental concerns. They are particularly crucial in islanded microgrid systems (<xref ref-type="bibr" rid="B14">Su et al., 2021</xref>), where they significantly contribute to maintaining the equilibrium between power supply and demand. However, DGs exhibit minimum operational power thresholds, leading to inefficiencies under low-load conditions. Consequently, optimizing the usage of diesel generators necessitates minimizing their operation at low loads while prioritizing their deployment for higher load demands. This strategy ensures efficient energy production and enhances the overall operational efficacy of microgrid systems, aligning with the objectives of balancing energy supply with demand while addressing the inherent limitations of DGs. The relationship between diesel generator fuel and power is given as follows.<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>Q</italic> is the fuel per hour of the diesel generator, <italic>P</italic> is the current power of the diesel generator, <italic>P</italic>
<sub>
<italic>0</italic>
</sub> is the rated power, <italic>&#x3b1;</italic> and <italic>&#x3b2;</italic> are the fuel consumption factors.</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Micro gas turbines</title>
<p>In the early stage of invention of micro gas turbine, due to the immaturity of related technology, which resulted in low power generation efficiency, the usage rate of micro gas turbine was not very high at first, but with the development of power generation technology, the power generation efficiency has been gradually improved, and the practicality has been enhanced after the size is reduced. As a kind of controllable distributed power generation unit, when the renewable energy power generation equipment becomes unstable due to the natural environment, the micro gas turbine can be adjusted to coordinate the microgrid to achieve the optimal operation state.</p>
<p>MGTs operate as rotary heat engines utilizing fuel and air, emerging as viable, energy-efficient, and eco-friendly power solutions suitable for urban, rural, and remote applications. The MGT system comprises components such as a gas bath wheel, combustion chamber, reheater, and compressor. The process involves air being drawn in and pressurized by the compressor, then preheated and mixed with fuel in the combustion chamber. The resultant high-temperature, high-pressure gas drives an electric motor to produce electricity. The output power of MGTs is directly proportional to fuel consumption, necessitating a mathematical model to optimize cost efficiency. This technological evolution underscores the MGT&#x2019;s significance in enhancing microgrid resilience and sustainability across diverse geographical locales. The magnitude of the generation power of the micro gas turbine is determined by the consumption of the fuel, and the mathematical model of its cost is as follows (<xref ref-type="bibr" rid="B4">Hosseini and Etemadi, 2008</xref>):<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x394;</mml:mo>
<mml:mi>t</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>V</mml:mi>
<mml:msub>
<mml:mi>&#x3b7;</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b7;</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.0753</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>65</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>3</mml:mn>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>0.3095</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>65</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>0.417</mml:mn>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>65</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>0.1068</mml:mn>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <italic>C</italic>
<sub>MT</sub> is the fuel cost of the micro gas turbine, <italic>C</italic> is the unit price of the fuel gas, <italic>P</italic>
<sub>MT</sub> is the power generated by the micro gas turbine at <italic>t</italic>, LHV is the low calorific value of natural gas.</p>
</sec>
<sec id="s2-2-3">
<title>2.2.3 Fuel cells</title>
<p>The power output of Fuel Cells (FCs) is directly correlated with the quantity of fuel supplied, allowing for modulation of power levels through the adjustment of fuel flow rates. FCs are characterized by their superior dynamic response capabilities, enabling rapid adjustments to power output in response to varying operational demands. This attribute not only enhances the adaptability of FCs within energy systems but also underscores their potential in applications requiring quick response times and flexible power generation.<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>fuel</mml:mtext>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mtext>cell</mml:mtext>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mtext>fuel</mml:mtext>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mtext>cell</mml:mtext>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mtext>fuel</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>fuel</mml:mtext>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mtext>cell</mml:mtext>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the output of the FCs, <italic>F</italic>
<sub>
<italic>Fuel</italic>
</sub> is the fuel flow rate and <italic>K</italic>
<sub>
<italic>fuel cell</italic>
</sub> is the conversion efficiency.</p>
</sec>
<sec id="s2-2-4">
<title>2.2.4 Distributed WT modelling</title>
<p>Wind Turbines (WTs) serve as devices for transforming airflow kinetic energy into mechanical energy, subsequently converted into electrical energy. The primary components of a WT include blades, a gearbox, and a generator. The process entails wind propelling the blades, thereby converting the wind&#x2019;s kinetic energy into mechanical energy. This mechanical energy is then amplified through gearbox-mediated blade acceleration, which powers the generator to convert mechanical energy into magnetic field energy, and ultimately into electrical energy. The correlation between a WT&#x2019;s output power and wind speed is mathematically represented in <xref ref-type="disp-formula" rid="e5">Equation 5</xref>, illustrating the efficiency of energy conversion under varying wind conditions:<disp-formula id="e5">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mtext>Wt</mml:mtext>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>v</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mtext>co</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mtext>co</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf2">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the output power of the fan at the time of <italic>t</italic>, <inline-formula id="inf3">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the rated output power of the fan, <inline-formula id="inf4">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the cut-in wind speed, <inline-formula id="inf5">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the rated wind speed, <inline-formula id="inf6">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the cut-out wind speed.</p>
</sec>
<sec id="s2-2-5">
<title>2.2.5 Distributed PV modelling</title>
<p>Photovoltaic (PV) power generation system is a new type of power generation model that uses the photovoltaic effect of semiconductor materials to directly convert solar radiation energy into electricity. The photovoltaic effect refers to the change of carrier distribution state and concentration of semiconductor materials after they are exposed to light, thus generating electric current and electric potential. Photovoltaic power generation system consists of photovoltaic panel module, controller, inverter and transformer, as shown in <xref ref-type="fig" rid="F2">Figures 2</xref>, <xref ref-type="fig" rid="F3">3</xref>, the photovoltaic panel module is easily affected by external conditions, making its output energy is not stable enough, the battery as an energy storage device, can be converted to solar panels to store solar energy, so as to meet the needs of continuous load operation. The temperature of the PV array at a certain moment is related to the current solar radiation intensity and the warming coefficient of the PV array, and the temperature of the PV array at a certain moment is given by the following equation.<disp-formula id="e6">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf7">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the temperature of the PV array at <italic>t</italic>, <inline-formula id="inf8">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the temperature of the PV array under the standard test conditions, <inline-formula id="inf9">
<mml:math id="m15">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the warming coefficient of the PV array, and <inline-formula id="inf10">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the intensity of solar radiation at <italic>t</italic>. The generation power of the PV power generation system is closely related to the current solar radiation intensity, the current temperature of the PV array and the temperature difference coefficient of the PV array, and its generation power expression is shown as follows.<disp-formula id="e7">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
<disp-formula id="e8">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf11">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the intensity of light radiation under standard test conditions, <inline-formula id="inf12">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the output electric power of PV at the time of <italic>t</italic>, <inline-formula id="inf13">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>T</mml:mi>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the output electric power of PV under standard test conditions, <italic>k</italic> denotes the temperature difference coefficient of PV array, and <italic>&#x3b8;</italic> denotes the radiation intensity coefficient.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Graph Neural Networks in SGAC framework.</p>
</caption>
<graphic xlink:href="fenrg-12-1384995-g002.tif"/>
</fig>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Results in case 1. <bold>(A)</bold> Frequency deviation <bold>(B)</bold> Total regulated output.</p>
</caption>
<graphic xlink:href="fenrg-12-1384995-g003.tif"/>
</fig>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Generation costs</title>
<p>The formulation of generation cost within the electricity production sphere is articulated via a sophisticated mathematical model, meticulously capturing the comprehensive economic obligations of power generation firms. This model is all-encompassing, amalgamating crucial operational cost elements integral to the generation process. It assimilates direct costs, such as fuel expenses encompassing a spectrum from fossil to renewable sources, and extends to cover the wide range of maintenance demands for generation infrastructure, encompassing routine checks, parts replacement, and emergency repairs. Furthermore, the model includes labor expenses covering wages, training, and health and safety measures for personnel, along with various indirect costs. These indirect expenses encompass regulatory compliance charges, environmental levies, and investments in technological innovation, essential for the electricity generation continuum.</p>
<p>Crafted with precision, the model reflects the intricate dynamics prevalent in the energy sector by integrating both variable costs, which alter with production levels and operational intensity, and fixed costs, which are invariant to output volume. This dual approach offers a comprehensive perspective on the economic terrain of electricity generation, covering the spectrum from initial capital outlay to incremental operating expenses.</p>
<p>Merging such varied economic components into a unified model provides stakeholders with an in-depth view of the critical economic considerations vital for prudent and sustainable power generation management. It facilitates a detailed comprehension of the interplay between different cost determinants and their collective influence on the cost-efficiency and -effectiveness of power plants. Ultimately, this model transcends being a mere cost inventory, evolving into a dynamic tool that underpins strategic decision-making and future-oriented planning, pivotal for the advancement of the power generation industry. The cost of power generation is as follows:<disp-formula id="e9">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>where <italic>P</italic>
<sub>
<italic>Gi</italic>
</sub> is the output of the <italic>ith</italic> unit, <italic>a</italic>
<sub>
<italic>i</italic>
</sub>
<italic>, b</italic>
<sub>
<italic>i</italic>
</sub>
<italic>,c</italic>
<sub>
<italic>i</italic>
</sub> are constants, and <italic>C</italic>
<sub>
<italic>i</italic>
</sub> is the cost of the <italic>ith</italic> unit.<disp-formula id="e10">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>actual</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>plan</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
<disp-formula id="e11">
<mml:math id="m24">
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>plan</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>plan</mml:mtext>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>plan</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where &#x394;<italic>P</italic>
<sub>
<italic>Gi</italic>
</sub> is the output of <italic>i</italic>th unit, <italic>P</italic>
<sub>
<italic>Gi</italic>,actual</sub> is the output of <italic>i</italic>th unit,<italic>&#x3b1;</italic>
<sub>
<italic>i</italic>
</sub>, <italic>&#x3b2;</italic>
<sub>
<italic>i</italic>
</sub>, <italic>&#x3b3;</italic>
<sub>
<italic>i</italic>
</sub> are coefficients.</p>
</sec>
<sec id="s2-4">
<title>2.4 Objective functions and constraints</title>
<p>The Differential Evolution Optimization for Load Frequency Control (DEO-LFC) framework represents a vanguard methodology in the domain of electrical grid management, specifically engineered to fortify frequency stability across power networks&#x2014;a critical factor for ensuring uninterrupted service and superior power quality within microgrid configurations. Achieving and maintaining an exact frequency balance is of paramount importance, as deviations from the established frequency spectrum can lead to detrimental effects, such as the deterioration of infrastructure, compromised quality of electricity, and a heightened risk of grid instability. Within the realm of microgrid management, the economic aspects of power generation take on a significant role, deeply influencing the operational dynamics and the economic viability of these systems. The implementation of efficacious frequency control measures is indispensable for reducing superfluous energy consumption and operational expenses, thereby enhancing the economic efficiency of microgrids, optimizing the use of resources, and improving the cost-efficiency of power generation initiatives.</p>
<p>Islanded microgrids, characterized by their compact scale and increased susceptibility to fluctuations in load demand, encounter unique challenges in achieving consistent frequency control. These standalone power systems require intricate and flexible management strategies that can effectively align the twin goals of cost minimization and optimization of system performance. The DEO-LFC approach addresses these challenges by deploying an innovative multi-objective optimization framework, carefully crafted to strike a balance between cost-effectiveness and dependable system performance. This framework is designed to minimize the adverse effects of operational constraints while preserving the integrity of economic and performance objectives, thus embodying a holistic strategy that caters to both economic and operational performance imperatives.</p>
<p>Incorporating multi-objective optimization techniques, the DEO-LFC methodology skillfully manages the intricate interplay between grid stability and economic factors in power generation. It delivers a refined solution that adeptly adjusts the equilibrium between ensuring grid stability and contemplating the economic dimensions of power generation. This versatile and comprehensive approach is uniquely suited to address the fluctuating demands of microgrid settings, guaranteeing frequency stability alongside a commitment to operational efficiency and fiscal judiciousness. The strategic formulation of objective functions and constraints under this methodology underscores its capacity to navigate the complexities of modern power systems, providing a robust framework for the sustainable and efficient management of energy resources in microgrids. Through the meticulous design of its optimization processes, the DEO-LFC strategy exemplifies an advanced paradigm in grid management, advocating for a harmonious integration of technical and economic considerations to foster resilient and economically viable microgrid ecosystems. The objective functions and constraints are as follows.<disp-formula id="e9a">
<mml:math id="m25">
<mml:mrow>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03B1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="italic">Gi</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi mathvariant="italic">Gi</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x03B3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(9a)</label>
</disp-formula>
<disp-formula id="e10a">
<mml:math id="m26">
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>in</mml:mtext>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>in</mml:mtext>
</mml:msubsup>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>min</mml:mi>
</mml:msubsup>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>in</mml:mtext>
</mml:msubsup>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>max</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>rate</mml:mtext>
</mml:msubsup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(10a)</label>
</disp-formula>where &#x394;<italic>P</italic>
<sub>order-&#x2211;</sub> is the total command, &#x394;<italic>P</italic>
<sub>
<italic>i</italic>
</sub>
<sup>max</sup> and &#x394;<italic>P</italic>
<sub>
<italic>i</italic>
</sub>
<sup>min</sup> are the limits of the <italic>ith</italic> unit, &#x394;<italic>P</italic>
<sub>
<italic>i</italic>
</sub>
<sup>in</sup> is the command of the <italic>ith</italic> unit.</p>
</sec>
</sec>
<sec id="s3">
<title>3 SGAC algorithm for DEO-LFC</title>
<p>In the realm of Artificial Intelligence (AI), rapid advancements have necessitated the adaptation to complex task deployment, with edge computing environments emerging as a pivotal solution due to their low latency, high throughput, and energy-efficient characteristics. These environments are increasingly applied across various sectors including intelligent Internet of Things (IoT), transportation, and healthcare, demanding efficient task deployment to maximize computing performance and resource allocation. Yet, task deployment poses a complex combinatorial optimization challenge, entangled with inter-task dependencies and multifaceted constraints. To navigate these complexities, the scholarly and industrial sectors have proposed innovative approaches, notably Graph Neural Networks (GNNs) and Deep Reinforcement Learning (DRL) methodologies.</p>
<p>GNNs offer a graphical framework to encapsulate inter-task dependencies, typically represented by a Directed Acyclic Graph (DAG) in LFC scenarios. The adoption of a graph structure transmutes the LFC challenge into a graph combinatorial optimization problem, thereby enhancing LFC&#x2019;s efficiency and robustness. GNNs, as a subset of artificial neural networks adept at processing graph data, can discern and manage task dependencies, facilitating superior task deployment outcomes.</p>
<p>Conversely, DRL, a subset of reinforcement learning that derives optimal strategies through environmental interaction, is instrumental in refining LFC strategies. DRL optimizes LFC policies to augment efficiency and precision, assimilating task dependencies and constraints to advance LFC performance.</p>
<p>This section delves into the synthesis of GNNs and DRL to tackle the LFC dilemma, particularly in isolated microgrid contexts. It envisages employing GNNs for delineating task interdependencies and optimizing these relationships. Concurrently, DRL will be leveraged to formulate and implement optimal LFC policies, aiming to address the intricate LFC issues inherent in islanded microgrids. This integrative approach signifies a promising direction for enhancing task deployment and LFC efficacy in complex computational landscapes.</p>
<sec id="s3-1">
<title>3.1 MDP modelling of DEO-LFCs</title>
<p>In Reinforcement Learning (RL), the dynamic interaction between an agent and its environment is conceptualized through a Markov Decision Process (MDP), serving as both the mathematical foundation and a key modeling instrument for RL challenges. An MDP framework typically encapsulates a state space, action space, state transition probabilities, and a reward function. Within this structure, the agent selects actions in accordance with the present state at each discrete time step, while the environment responds by presenting a subsequent state and associated reward, as dictated by the state transition probability function and the reward function. The MDP framework posits that state transitions in the decision-making process adhere to the Markov property&#x2014;meaning the forthcoming state is contingent solely on the current state and the executed action, devoid of any historical influence. The quintessential components of an MDP include.<list list-type="simple">
<list-item>
<p>1) State space: the set of all possible states.</p>
</list-item>
<list-item>
<p>2) Action space: The set of all possible actions.</p>
</list-item>
<list-item>
<p>3) State transfer probability: describes the probability distribution from one state and one action to the next state.</p>
</list-item>
<list-item>
<p>4) Reward function: describes the immediate reward obtained after performing an action in a state.</p>
</list-item>
</list>
</p>
<p>In MDP, the goal of an agent is to find an optimal strategy, i.e., to choose an optimal action in each state to maximize the expected cumulative reward.</p>
<sec id="s3-1-1">
<title>3.1.1 Action space</title>
<p>In the sphere of sophisticated power grid management, the imperative for an advanced control system is pronounced. Such a system is essential for the generation and distribution of precise control directives to every unit within specified sectors, highlighting the need for a comprehensive action space for the supervisory entity. This action space is crucial, crafted to encompass a full array of commands vital for the seamless operation of each unit. The contemplated action space is intricate, mirroring the wide array of decisions the controlling agent must implement. These decisions span various operational aspects, from adjusting power output levels to fine-tuning for system equilibrium and reliability. The complexity inherent in this space reflects the diverse nature of the required tasks, emphasizing the necessity for the agent to exhibit exceptional precision and adaptability.</p>
<p>Furthermore, the action space illustrates the complex coordination and synergy required among different units to attain collective operational efficacy. It establishes a structure that not only facilitates task execution at the individual unit level but also integrates these activities within the broader grid management goals. This integration demands a degree of interaction and cooperation that goes beyond mere directive issuance, necessitating a unified approach that aligns with overarching performance objectives.</p>
<p>Thus, the agent must adeptly navigate this action space, informed by the dynamic interrelations within the power grid, to make decisions that are both cognizant of the current context and anticipatory of future grid conditions. Such advanced decision-making capability is crucial for ensuring optimal grid performance, reducing operational interruptions, and enabling a resilient adaptation to fluctuating demand and supply scenarios. The meticulously designed action space is a fundamental element of the control architecture, endowed with the complexity and strategic insight necessary to meet the rigorous demands of contemporary grid operation and management. The action space is as follows:<disp-formula id="e11a">
<mml:math id="m27">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(11a)</label>
</disp-formula>where <inline-formula id="inf14">
<mml:math id="m28">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the total command.</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 State space</title>
<p>The autonomous control agent plays a pivotal role in diligently managing a comprehensive database of operational metrics for the standalone microgrid. Its core function is to meticulously implement decisions that adjust for frequency deviations, leveraging an extensive collection of real-time and historical data. This role is crucial for the continuous monitoring and adjustment of the power output from each turbine unit, especially critical in environments lacking rapid-response mechanisms to counteract significant power fluctuations.</p>
<p>Ultimately, this meticulous and strategic methodology endows the autonomous agent with the capabilities required to uphold the operational integrity of the microgrid. It underscores the agent&#x2019;s critical contribution to maintaining the resilience, efficiency, and stability of the energy system, navigating the intricate dynamics of managing standalone power grids effectively. The state space is as follows:<disp-formula id="e12">
<mml:math id="m29">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mi>t</mml:mi>
</mml:msubsup>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>t</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>where <inline-formula id="inf15">
<mml:math id="m30">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the total output.</p>
</sec>
<sec id="s3-1-3">
<title>3.1.3 Reward functions</title>
<p>Within the domain of power system operational efficiency optimization, reinforcement learning algorithms frequently utilize two paramount metrics as reward functions: frequency deviation and generation cost. These metrics are integral for assessing system performance and economic viability, respectively. To bolster the training efficacy and mitigate the risk of frequency tuning errors during the exploration phase, a penalty factor is strategically implemented. This factor is aimed at accelerating the learning curve by imposing penalties for actions resulting in non-ideal outcomes, such as deviations from the desired frequency levels.</p>
<p>The incorporation of a penalty factor serves to guide the learning algorithm towards optimal actions by introducing a cost for inaccuracies, thereby enhancing the training process&#x2019;s efficiency and dependability. This approach addresses the exploration-exploitation dilemma in reinforcement learning, necessitating a balance between investigating novel actions and leveraging established strategies. This equilibrium is crucial, especially in intricate systems where the ramifications of less-than-optimal decisions can significantly impact system stability and operational expenses.</p>
<p>By embedding a penalty factor focused on rectifying frequency tuning discrepancies, the training methodology is refined to emphasize system stability and cost efficiency. Consequently, this adjustment improves the power system&#x2019;s operational performance, aligning with the objectives of maintaining system reliability and economic efficiency. The reward is as follows:<disp-formula id="e13">
<mml:math id="m31">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn mathvariant="italic">2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn mathvariant="italic">3</mml:mn>
</mml:msub>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m32">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0.05</mml:mn>
<mml:mi>H</mml:mi>
<mml:mi>Z</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>10</mml:mn>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0.05</mml:mn>
<mml:mi>H</mml:mi>
<mml:mi>Z</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>where <italic>r</italic> is the reward and <italic>C</italic>
<sub>
<italic>i</italic>
</sub> is the punishment function.</p>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 SGAC algorithm-based DEO-LFC application</title>
<sec id="s3-2-1">
<title>3.2.1 Maximum entropy exploration strategy</title>
<p>Upon reviewing the trio of DRL algorithms delineated in the preceding section, it becomes apparent that the challenges confronting the DRL domain are fundamentally consistent. These challenges include the intricacies of exploration and decision-making within high-dimensional state spaces, alongside the convergence dilemmas encountered in function optimization. The former challenge is attributed to the exponential growth in the number of states within extensive state spaces, which significantly escalates computational time and resource allocation, thereby complicating the identification of optimal policies within constrained temporal and spatial parameters. The latter challenge pertains to the non-convex optimization issues inherent in algorithmic exploration and decision-making processes. This complication arises from the prevalence of numerous local optima within the deep neural network&#x2019;s parameter space, impeding the algorithm&#x2019;s progression towards a global optimum due to entrapment in local optima during optimization phases.</p>
<p>It is crucial to acknowledge that these two predominant challenges are not mutually exclusive but are interlinked, necessitating concurrent resolution. Thus, addressing these issues collectively is paramount. This paper introduces a novel approach through the development of a flexible actor-critic algorithm, which leverages the maximum entropy framework, diverging from traditional DRL algorithms that solely prioritize maximizing long-term rewards. The Soft Actor-Critic (SAC) algorithm innovates by incorporating an action&#x2019;s maximum entropy estimation into its action selection strategy, aiming to enhance decision-making robustness and algorithmic convergence. The adoption of a maximum entropy-based strategy for the objective function, as depicted in <xref ref-type="disp-formula" rid="e15">Equation 15</xref>, signifies a strategic pivot designed to mitigate the aforementioned challenges by promoting a more explorative and globally informed optimization process.<disp-formula id="e15">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>where <italic>&#x3b1;</italic> denotes the temperature control factor, which is used to regulate the importance of the entropy term <inline-formula id="inf16">
<mml:math id="m34">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> relative to the reward term <inline-formula id="inf17">
<mml:math id="m35">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Incorporating the maximum entropy function into the Soft Actor-Critic (SAC) algorithm fundamentally alters the probabilistic landscape of action selection. This approach ensures a distribution mechanism that mitigates the likelihood of the agent persistently favoring actions with disproportionately high probabilities. The primary advantage of integrating the maximum entropy principle lies in its capacity to randomize the strategic optimization pathway. This randomness acts as a catalyst for enhanced exploration during the initial training phases, enabling the algorithm to evaluate and learn from a broader spectrum of action outcomes. Such a mechanism not only accelerates the training process by enriching the exploration domain but also prevents the convergence on suboptimal policies in later stages by discouraging repetitive action selection.</p>
<p>By diminishing the repetitive selection of identical actions, the strategy effectively minimizes the perturbation induced by noise, thereby streamlining the algorithm&#x2019;s path to convergence. This reduction in noise influence is crucial for achieving a more stable and efficient learning trajectory. The strategic application of the maximum entropy function, therefore, plays a critical role in balancing exploration with exploitation, optimizing training velocity, and facilitating smoother algorithmic convergence by alleviating the impact of stochastic behaviors on the learning process.</p>
<p>In order to make the algorithm work in the continuous domain, the SAC algorithm sets up function approximators for the value function and the policy function that can assist the function to be updated for optimization. There are three main types of objective optimization functions in the SAC algorithm, namely, the state value function (<inline-formula id="inf18">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>&#x3c8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>) controlled by the parameter <italic>&#x3c8;</italic>, the action value function (<inline-formula id="inf19">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>) controlled by the parameter <inline-formula id="inf20">
<mml:math id="m38">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and the policy function (<inline-formula id="inf21">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>) controlled by the parameter <italic>&#x3b8;</italic>. The function approximator is used to perform the gradient of the three objective functions. The function approximator&#x2019;s role is to update the gradient of the three types of objective functions, and the specific updating formulas are shown in <xref ref-type="disp-formula" rid="e16">Equations 16</xref>&#x2013;<xref ref-type="disp-formula" rid="e18">18</xref>, respectively.<disp-formula id="e16">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mo>&#x2207;</mml:mo>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>&#x3c8;</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>V</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>&#x3c8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>&#x3c8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="e17">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mo>&#x2207;</mml:mo>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mover accent="true">
<mml:mi>&#x3c8;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
<disp-formula id="e18">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mo>&#x2207;</mml:mo>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3f5;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>where <inline-formula id="inf22">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mover accent="true">
<mml:mi>&#x3c8;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the target state value function, the stability of the state value function is controlled by smoothing the network parameters <inline-formula id="inf23">
<mml:math id="m44">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3c8;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> ; <inline-formula id="inf24">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the policy function with the addition of the noise vector <inline-formula id="inf25">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3f5;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which converts the random sampling process to input random noise through the reparameterization technique, so that the function can undergo gradient updating.</p>
<p>In order to avoid over-estimation of the Q values of certain actions by the action value function, the SGAC algorithm adopts the cropping dual Q value learning technique, i.e., it uses two identical Q networks, Q1 and Q2, and achieves the purpose of reducing the training bias and improving the stability and robustness of the algorithm by selecting the smaller of the largest Q values obtained from the Q1 and Q2 networks. In order to simplify the block diagram structure. For the current Q network, due to the deletion of the V network, the gradient update formula of the corresponding action value function <inline-formula id="inf26">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> will be rewritten without the state value function, as shown in <xref ref-type="disp-formula" rid="e19">Equation 19</xref>.<disp-formula id="e19">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mo>&#x2207;</mml:mo>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mover accent="true">
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>where <inline-formula id="inf27">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mover accent="true">
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the target action value function, and <inline-formula id="inf28">
<mml:math id="m50">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the optimal value of the temperature control factor <italic>&#x3b1;</italic> at the current moment, which can be expressed by <xref ref-type="disp-formula" rid="e20">Equation 20</xref> containing the minimum entropy constant <inline-formula id="inf29">
<mml:math id="m51">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>H</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>, i.e., the opposite of the action space dimension.<disp-formula id="e20">
<mml:math id="m52">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mi>argmin</mml:mi>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:munder>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:msubsup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mover accent="true">
<mml:mi>H</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(20)</label>
</disp-formula>
</p>
<p>In the deep reinforcement learning-based LFC scenario studied in this paper, the training process of the recommended intelligences model can also be represented as a serialisation process based on a Markov decision process.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Actor networks</title>
<p>For the policy function of the system, the policy gradient algorithm is particularly suitable for long-term interactive recommendation scenarios because of its iterative optimization feature, which can gradually accumulate experience through the continuous interaction between the agent and the environment. In this paper, we use the classic algorithm in the policy gradient algorithm to optimise the policy, and the policy function of this algorithm is shown below.<disp-formula id="e21">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msup>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(21)</label>
</disp-formula>
</p>
<p>Among them, <italic>B</italic> denotes the experience pool, which is used to store the current state, target state, action and the instant reward obtained after interacting with the environment of each recommending agent, <inline-formula id="inf30">
<mml:math id="m54">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the instant reward obtained by the agent from the user&#x2019;s feedback at the time of <italic>t</italic>, and the attenuation factor of the sub-scenario of t. <italic>&#x3b3;</italic> is used to balance the relationship between the instant reward and the delayed reward. Through the cumulative effect of the delayed reward and the instant reward, the cumulative reward of the application in the long-term recommending scenario can be computed.</p>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Network of critics</title>
<p>This study introduces an advanced policy gradient algorithm featuring a self-adjusting temperature control factor, designed to enhance system robustness against interference and to accelerate convergence rates. A distinctive aspect of this approach is the incorporation of an entropy-based action selection term into the policy objective function, as detailed in <xref ref-type="disp-formula" rid="e21">Equation 21</xref>. This modification aims to optimize the policy function&#x2019;s performance by leveraging the value function for reward assessment, thereby minimizing bias throughout the training phase. The formulation of the actor component&#x2019;s policy function within the SGAC algorithm is elucidated in <xref ref-type="disp-formula" rid="e22">Equation 22</xref>. This innovative mechanism facilitates a more dynamic adaptation process, significantly improving the algorithm&#x2019;s efficiency in navigating complex environments and achieving optimal decision-making strategies.<disp-formula id="e22">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mi>E</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(22)</label>
</disp-formula>where <inline-formula id="inf31">
<mml:math id="m56">
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the entropy term regulated by the self-updating temperature-control factor <inline-formula id="inf32">
<mml:math id="m57">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf33">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the reward term evaluated using the action-value function of the critics&#x2019; section, and <inline-formula id="inf34">
<mml:math id="m59">
<mml:mrow>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> denotes all possible actions predicted by the current strategy function.</p>
<p>Deep reinforcement learning algorithms based on the Actor-Critic architecture usually choose the state value function or action value function as an important basis for policy optimisation. In the SAC algorithm, the action value function, as a direct influence on the policy function update, is particularly important in the design. The algorithm adopted in this paper contains one actor network and four critic networks. The actor network is a strategy network, while the critic network contains two identical current action value networks and two identical target action value networks. The structure of the current action value network and the target action value network is basically the same, the only difference is that the parameter <inline-formula id="inf35">
<mml:math id="m60">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> in the target action value network is realised through the smoothing factor <inline-formula id="inf36">
<mml:math id="m61">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to adjust its own parameter and the parameter in the current action value network, and the specific formula of the current action value function is shown in <xref ref-type="disp-formula" rid="e23">Equation 23</xref>.<disp-formula id="e23">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>Q</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(23)</label>
</disp-formula>
</p>
<p>As can be seen from <xref ref-type="disp-formula" rid="e23">Equation 23</xref>, the current action value function in the SGAC algorithm contains two terms related to the value of Q. <inline-formula id="inf37">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the predicted action value term, whose main role is to directly participate in the evaluation operation of some of the actors&#x2019; strategy functions, so as to achieve the guidance for the optimization and updating of the strategy network. <inline-formula id="inf38">
<mml:math id="m64">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>Q</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the real action value term based on the reward and goal state values, which is shown in <xref ref-type="disp-formula" rid="e24">Equation 24</xref>.<disp-formula id="e24">
<mml:math id="m65">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>Q</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mover accent="true">
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(24)</label>
</disp-formula>
</p>
<p>Since the SGAC algorithm deletes the state value network, the target state value is represented as a target action value term containing a decay factor and an entropy term containing a self-renewal temperature control factor. The parameter updating method of the target action value function is a flexible updating method based on the smoothing factor <italic>&#x3c4;</italic>, and the specific updating formula is shown in <xref ref-type="disp-formula" rid="e25">Equation 25</xref>.<disp-formula id="e25">
<mml:math id="m66">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
<label>(25)</label>
</disp-formula>
</p>
<p>From <xref ref-type="disp-formula" rid="e25">Equation 25</xref>, it can be seen that the real action value function including the target state value term is a major innovation of the SGAC algorithm compared with the traditional Q-learning algorithm, i.e., the introduction of the operation term based on the entropy of the action selection is used to balance the relationship between exploration and exploitation, so as to improve the stability of the algorithm and the convergence speed. In addition, the current action value network and the target action value network in the SGAC algorithm contain two identical network structures, and the reward evaluation of the strategy function is performed by selecting a smaller Q value from the same network structure, which effectively reduces the training bias caused by overestimation.</p>
<p>The SGAC algorithm has a significant improvement over the original SAC algorithm in terms of updating the temperature control factors. This improvement is mainly reflected in the fact that the SGAC algorithm uses the constrained optimisation method to split the entropy term into the strategy entropy related to the strategy and the minimum entropy unrelated to the strategy. The self-renewal temperature control factor <inline-formula id="inf39">
<mml:math id="m67">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> can be expressed as a temperature control network with the temperature control factor as the network parameter, and the specific formula is shown in <xref ref-type="disp-formula" rid="e26">Equation 26</xref>.<disp-formula id="e26">
<mml:math id="m68">
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>B</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mover accent="true">
<mml:mi>H</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(26)</label>
</disp-formula>
</p>
<p>Compared with the original SAC algorithm, the self-updating temperature control factor <inline-formula id="inf40">
<mml:math id="m69">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> solves the problem of maintaining the weights of the quotient terms unchanged in the original SAC algorithm, which makes the algorithm more intelligent in utilising the entropy terms, and thus can show better performance in complex learning tasks.</p>
</sec>
<sec id="s3-2-4">
<title>3.2.4 Graph neural networks</title>
<p>The GRL algorithm proposed in this paper consists of graph convolutional neural network (GCN) and deep deterministic policy gradient (DDPG), which includes graph policy network and graph value network, its overall structure is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. GRL algorithm consists of graph policy network and graph value network.<list list-type="simple">
<list-item>
<p>1) The input of the graph strategy network is the state diagram of an islanded microgrid considering the knowledge of strong nonlinear currents, which contains the adjacency matrix <inline-formula id="inf41">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the feature matrix <inline-formula id="inf42">
<mml:math id="m71">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold">X</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mi>t</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> with the information of frequency state, power, etc., and the output is the output of the equipment power.</p>
</list-item>
<list-item>
<p>2) The initial input of the graph value network is the same as that of the graph strategy network, the input of the first fully connected layer (FC) after the graph convolution layer is composed of the features output from the graph convolution layer and the actions generated by the graph strategy network, and the final output of the graph value network is a one-dimensional data, i.e., the value of <italic>Q</italic> of the actions applied in the environment state, which is used to evaluate the actions.</p>
</list-item>
</list>
</p>
<p>In the graph strategy network <inline-formula id="inf43">
<mml:math id="m72">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and graph value network <inline-formula id="inf44">
<mml:math id="m73">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> , the node feature information flows through the real topological layers of the islanded microgrid in the form of hidden features, and the propagation feature <inline-formula id="inf45">
<mml:math id="m74">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> at layer <inline-formula id="inf46">
<mml:math id="m75">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> can be expressed as below.<disp-formula id="e27">
<mml:math id="m76">
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:msup>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mi mathvariant="bold">W</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi mathvariant="bold">D</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold">E</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="bold">D</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(27)</label>
</disp-formula>where <inline-formula id="inf47">
<mml:math id="m77">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the input features of layer <inline-formula id="inf48">
<mml:math id="m78">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> ; <inline-formula id="inf49">
<mml:math id="m79">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">W</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the weight matrix of layer <inline-formula id="inf50">
<mml:math id="m80">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> ; <inline-formula id="inf51">
<mml:math id="m81">
<mml:mrow>
<mml:mi mathvariant="bold">D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the degree matrix of the adjacency matrix <inline-formula id="inf52">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> ; <inline-formula id="inf53">
<mml:math id="m83">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the activation function; <inline-formula id="inf54">
<mml:math id="m84">
<mml:mrow>
<mml:mi mathvariant="bold">E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the unit matrix.</p>
<p>Therefore, the rules for transferring feature information between layers of the GRL algorithm are shown in <xref ref-type="disp-formula" rid="e28">Equations 28</xref>, <xref ref-type="disp-formula" rid="e29">29</xref>.<disp-formula id="e28">
<mml:math id="m85">
<mml:mrow>
<mml:mfenced open="{" close="" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Relu</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">D</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mi mathvariant="bold">W</mml:mi>
<mml:mtext>GCN</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold">b</mml:mi>
<mml:mtext>GCN</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mover accent="true">
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold">I</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(28)</label>
</disp-formula>
<disp-formula id="e29">
<mml:math id="m86">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Relu</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mi mathvariant="bold">W</mml:mi>
<mml:mtext>FC</mml:mtext>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">b</mml:mi>
<mml:mtext>FC</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(29)</label>
</disp-formula>where <inline-formula id="inf55">
<mml:math id="m87">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold">W</mml:mi>
<mml:mtext>GCN</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf56">
<mml:math id="m88">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="bold">b</mml:mi>
<mml:mtext>GCN</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the weight matrix and bias vector of the <italic>l</italic> layer GCN respectively; Relu(.) is the activation function; <inline-formula id="inf57">
<mml:math id="m89">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">W</mml:mi>
<mml:mtext>FC</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf58">
<mml:math id="m90">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">b</mml:mi>
<mml:mtext>FC</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the weight matrix and bias vector of the fully connected layer respectively. The graph value network optimizes the parameters by minimising the loss function as shown in <xref ref-type="disp-formula" rid="e30">Equations 30</xref>, <xref ref-type="disp-formula" rid="e31">31</xref>.<disp-formula id="e30">
<mml:math id="m91">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>Q</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold">X</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mi>t</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold">a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(30)</label>
</disp-formula>
<disp-formula id="e31">
<mml:math id="m92">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold">X</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mi>t</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold">X</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mi>t</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2223;</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2223;</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(31)</label>
</disp-formula>where <inline-formula id="inf59">
<mml:math id="m93">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold">X</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mi>t</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the state vector associated with the graph adjacency matrix and graph identity matrix; <inline-formula id="inf60">
<mml:math id="m94">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the target Q value; <inline-formula id="inf61">
<mml:math id="m95">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the expectation function; <inline-formula id="inf62">
<mml:math id="m96">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the value network under the parameter <inline-formula id="inf63">
<mml:math id="m97">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> ; <inline-formula id="inf64">
<mml:math id="m98">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the action vector, <inline-formula id="inf65">
<mml:math id="m99">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold">s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="bold">X</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mi>t</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2223;</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>; <inline-formula id="inf66">
<mml:math id="m100">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the immediate reward; <italic>&#x3b3;</italic> is the discount factor; <inline-formula id="inf67">
<mml:math id="m101">
<mml:mrow>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the target Q value of the target graph value network under the parameter <inline-formula id="inf68">
<mml:math id="m102">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>; <inline-formula id="inf69">
<mml:math id="m103">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the target strategy value of the target graph strategy network under the parameter <inline-formula id="inf70">
<mml:math id="m104">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Case studies</title>
<p>Within the ambit of this research, the efficacy of the DEO-LFC architecture, employing the SGAC algorithm, underwent a stringent assessment against a backdrop of a sophisticated CSG microgrid LFC paradigm, as explicated in the seminal work of (<xref ref-type="bibr" rid="B24">Yousef et al., 2014</xref>), with the deployment of parameters extrapolated from verifiable empirical datasets as expounded in (<xref ref-type="bibr" rid="B1">Bengiamin and Chan, 1982</xref>). The microgrid subject to this study operates at a nominal voltage of 10&#xa0;kV and integrates a multifaceted energy portfolio, including a 1.04 MWp solar photovoltaic array, a 50&#xa0;kW wind power installation, a 1220&#xa0;kW diesel power generator, a 2000&#xa0;kWh energy storage system, and a 300&#xa0;kW facility for electric This eclectic mix of energy sources and storage capabilities facilitates a fluid and efficacious transition across various strata of grid integration, from micro-energy to energy storage. This eclectic mix of energy sources and storage capabilities facilitates a fluid and efficacious transition across various strata of grid integration, from micro-energy production and load management to the incorporation of renewable energy sources and the execution of energy control systems both locally and remotely.</p>
<p>The analytic scrutiny of the DEO-LFC model, predicated on the SGAC algorithm, was conducted in juxtaposition with a spectrum of alternative This encompassed DEO-LFC frameworks predicated on algorithms such as Soft Q-Learning (<xref ref-type="bibr" rid="B10">Mi et al., 2016</xref>), Proximal Policy Optimisation (PPO) (<xref ref-type="bibr" rid="B2">Chen et al., 1991</xref>), and the SGAC algorithm (<xref ref-type="bibr" rid="B16">Wang et al., 1993</xref>). Proximal Policy Optimisation (PPO) (<xref ref-type="bibr" rid="B23">Yan et al., 2022</xref>), Trust Region Policy Optimization (TRPO) (<xref ref-type="bibr" rid="B9">Mahboob Ul Hassan et al., 2022</xref>), Distributed Distributional Deterministic Policy Gradients (D4PG) (<xref ref-type="bibr" rid="B27">Yu et al., 2012</xref>), Asynchronous Actor-Critic Agents (A3C) (<xref ref-type="bibr" rid="B25">Yu et al., 2015</xref>), Twin Delayed Deep Deterministic Policy Gradient (TD3) (<xref ref-type="bibr" rid="B13">Shangguan et al., 2021</xref>), Deep Deterministic Policy Gradient (DDPG) (<xref ref-type="bibr" rid="B7">Li et al., 2021</xref>), Double Deep Q-Network (DDQN) (<xref ref-type="bibr" rid="B3">Chen et al., 2022</xref>), Deep Q-Network (DQN) (<xref ref-type="bibr" rid="B29">Zhang et al., 2023b</xref>), Distributed Model Predictive Control (DMPC) (<xref ref-type="bibr" rid="B14">Su et al., 2021</xref>), Model Predictive Control (MPC) (<xref ref-type="bibr" rid="B4">Hosseini and Etemadi, 2008</xref>), Fuzzy Fractional Order Proportional Integral (Fuzzy-FOPI) (<xref ref-type="bibr" rid="B15">Wang et al., 2013</xref>), Fuzzy Proportional Integral (Fuzzy-PI) (<xref ref-type="bibr" rid="B3">Chen et al., 2022</xref>), and Particle Swarm Optimisation Proportional Integral (PSO-PI) (<xref ref-type="bibr" rid="B12">Peng et al., 2023</xref>) for LFC purposes.</p>
<p>This extensive comparative review was meticulously designed to evaluate the DEO-LFC framework, undergirded by the SGAC algorithm, in terms of efficiency, reliability, and adaptability within the operational milieu of advanced microgrid systems. The evaluative criteria were centred on the system&#x2019;s proficiency in maintaining voltage stability, enhancing the integration and The evaluative criteria were centred on the system&#x2019;s proficiency in maintaining voltage stability, enhancing the integration and exploitation of renewable energy resources, securing The evaluative criteria were centred on the system&#x2019;s proficiency in maintaining voltage stability, enhancing the integration and exploitation of renewable energy resources, securing dependable energy storage and retrieval mechanisms, and orchestrating efficacious load management protocols. Research aims to shed light on the transformative potential of cutting-edge deep learning algorithms in augmenting the operational efficiency of smart This, in turn, is envisaged to catalyze the evolution towards energy infrastructures that are not only more sustainable but also markedly more resilient.</p>
<sec id="s4-1">
<title>4.1 Case 1: random disturbance</title>
<p>In the present investigation, step disturbances were systematically introduced into the Case study to meticulously evaluate the system&#x2019;s response and resilience under perturbed conditions. The outcomes of this experimental setup are comprehensively documented and presented through a series of visual representations and quantitative data analyses, spanning <xref ref-type="fig" rid="F4">Figure 4</xref>. The outcomes of this experimental setup are comprehensively documented and presented through a series of visual representations and quantitative data analyses, spanning <xref ref-type="fig" rid="F4">Figure 4</xref> and including the detailed numerical results compiled in <xref ref-type="table" rid="T1">Table 1</xref>. This approach was deliberately chosen to facilitate a nuanced understanding of the system&#x2019;s dynamics and its capability to maintain stability or adapt to sudden changes in operating conditions. This approach was deliberately chosen to facilitate a nuanced understanding of the system&#x2019;s dynamics and its capability to maintain stability or adapt to sudden changes in operating conditions.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Results in case 2.</p>
</caption>
<graphic xlink:href="fenrg-12-1384995-g004.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Data of case 1.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Control algorithms</th>
<th align="center">Average frequency error (Hz)</th>
<th align="center">Generation cost ($)</th>
</tr>
<tr>
<th align="center">&#x7c;&#x394;f <italic>&#x7c;</italic>avg</th>
<th align="center">C<sup>total</sup>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SGAC</td>
<td align="center">0.00980</td>
<td align="center">3577.18</td>
</tr>
<tr>
<td align="center">soft Q-learning</td>
<td align="center">0.01169</td>
<td align="center">3580.87</td>
</tr>
<tr>
<td align="center">PPO</td>
<td align="center">0.01315</td>
<td align="center">3580.96</td>
</tr>
<tr>
<td align="center">TRPO</td>
<td align="center">0.01046</td>
<td align="center">3580.78</td>
</tr>
<tr>
<td align="center">D4PG</td>
<td align="center">0.01167</td>
<td align="center">3580.60</td>
</tr>
<tr>
<td align="center">A3C</td>
<td align="center">0.01076</td>
<td align="center">3580.66</td>
</tr>
<tr>
<td align="center">TD3</td>
<td align="center">0.01121</td>
<td align="center">3580.48</td>
</tr>
<tr>
<td align="center">DDPG</td>
<td align="center">0.01215</td>
<td align="center">3580.42</td>
</tr>
<tr>
<td align="center">DDQN</td>
<td align="center">0.01213</td>
<td align="center">3580.51</td>
</tr>
<tr>
<td align="center">DQN</td>
<td align="center">0.01458</td>
<td align="center">3580.15</td>
</tr>
<tr>
<td align="center">DMPC</td>
<td align="center">0.01353</td>
<td align="center">3580.24</td>
</tr>
<tr>
<td align="center">MPC</td>
<td align="center">0.01227</td>
<td align="center">3580.42</td>
</tr>
<tr>
<td align="center">Fuzzy-FOPI</td>
<td align="center">0.02603</td>
<td align="center">3578.26</td>
</tr>
<tr>
<td align="center">Fuzzy-PI</td>
<td align="center">0.03247</td>
<td align="center">3580.15</td>
</tr>
<tr>
<td align="center">PSO-PI</td>
<td align="center">0.02531</td>
<td align="center">3578.38</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The inclusion of step disturbances serves as a critical methodological tool to simulate real-world operational challenges, providing insights into The inclusion of step disturbances serves as a critical methodological tool to simulate real-world operational challenges, providing insights into the system&#x2019;s robustness and the efficacy of implemented control strategies. This structured presentation of findings, leveraging both graphical and This structured presentation of findings, leveraging both graphical and tabular formats, is designed to provide a comprehensive overview of the experimental results, fostering an in-depth analysis of the system&#x2019;s response patterns. The empirical evidence gathered through this methodology is instrumental in validating the theoretical models and hypotheses posited in the study, thereby contributing to the advancement of the system. The empirical evidence gathered through this methodology is instrumental in validating the theoretical models and hypotheses posited in the study, thereby contributing to the advancement of knowledge in the field. Furthermore, the detailed exposition of results in this manner adheres to rigorous scientific communication standards, ensuring clarity, precision, and replicability of the research findings.</p>
<p>
<xref ref-type="table" rid="T1">Table 1</xref>&#x2019;s analysis provides a detailed comparison of the SGAC algorithm against other algorithmic models, focusing on frequency deviation and generation cost metrics. The findings reveal that the frequency deviations with other strategies were 1.089&#x2013;4.155 times greater than with the SGAC algorithm. Additionally, the SGAC algorithm demonstrated a reduction in generation costs by 0.31%&#x2013;1.28% over its counterparts, underscoring its superior efficiency and control in microgrid management. A deeper investigation into frequency response and diesel generator outputs across different control strategies highlights the variance in performance and the efficacy of each control mechanism. The SGAC algorithm emerges as the top performer, with soft Q-learning noted as a strong alternative.</p>
<p>The standout performance of both the SGAC and soft Q-learning algorithms is linked to their use of maximum entropy exploration mechanisms, enabling precise adjustments in learning rates and importance weighting through an updated experience-sharing framework. This adaptability allows for customized control strategies in various zones, enhancing operational flexibility. Particularly, the SGAC algorithm excels in making decisions based on dynamic joint trajectories and historical data, bypassing traditional policy evaluation methods and improving its responsiveness to learning adjustments.</p>
<p>The SGAC algorithm&#x2019;s adaptability and control effectiveness across diverse system conditions firmly establish its role as a leader in the field of reinforcement learning, distinguished by its straightforward and universally applicable parameters. However, applying reinforcement learning broadly faces challenges, such as setting a shared exploration goal for multiple agents in complex tasks and dealing with the instability from agents needing to respond to each other.</p>
<p>The introduction of multi-agent reinforcement learning approaches, focusing on collective characteristics, marks a significant advancement in overcoming these challenges, steering reinforcement learning towards achieving dynamic tasks through autonomous decision-making and agent exploration.</p>
<p>Further examination of operational dynamics, as shown in <xref ref-type="fig" rid="F3">Figure 3B</xref>, reveals the system&#x2019;s ability to closely follow load disturbances, including negative and square wave perturbations. The LFC units&#x2019; power outputs adjust to address unpredictable power changes effectively. An in-depth look at the regulation curves of various LFC units in <xref ref-type="fig" rid="F3">Figure 3A</xref> shows a strategic allocation based on regulation costs and disturbance types, leading to optimized frequency control. The uniform micro-increment rate principle guides power distribution among LFC units, resulting in an economically efficient power output, contrasting with other DRL models that lack refinement mechanisms and heavily rely on theoretical models, thus limiting their control accuracy.</p>
</sec>
<sec id="s4-2">
<title>4.2 Case 2: step disturbance and renewable disturbance</title>
<p>In the research delineated within this document, a sophisticated smart distribution grid model, replete with a plethora of renewable energy sources, is meticulously constructed to facilitate an in-depth analysis of the SGAC algorithm&#x2019;s operational performance amidst a highly stochastic environment. The model is enriched with an array of renewable energy sources, encompassing electric vehicles, wind turbines, small-scale hydropower plants, micro-gas turbines, fuel cells, photovoltaic systems, and other energy sources. The model is enriched with an array of renewable energy sources, encompassing electric vehicles, wind turbines, small-scale hydropower plants, micro gas turbines, fuel cells, photovoltaic systems, and biomass energy solutions. Notably, certain renewable energy sources such as electric vehicles, wind power, and photovoltaic power generation exhibit considerable variability and unpredictability in their output. Consequently, these sources are modelled as sources of random load disturbances, their power outputs being incorporated into the system without contributing to the frequency Consequently, these sources are modelled as sources of random load disturbances, their power outputs being incorporated into the system without contributing to the frequency regulation mechanisms.</p>
<p>The variability inherent in wind power generation is captured through the application of finite bandwidth white noise as the input signal for the wind turbine model, thereby replicating the stochastic nature of wind speeds. The variability inherent in wind power generation is captured through the application of finite bandwidth white noise as the input signal for the wind turbine model, thereby replicating the stochastic nature of wind speeds. Similarly, the active power output for the photovoltaic generation model is derived by emulating the diurnal variations in solar irradiance. Characteristic of these renewable energy sources, thereby providing a robust framework for evaluating the efficacy of the SGAC algorithm under conditions that closely mimic real-world operational challenges.</p>
<p>Comprehensive details pertaining to the specifications and operational parameters of each energy unit incorporated within this model can be found in This extensive cataloguing of parameters is instrumental in underpinning the simulation exercises with a high degree of accuracy and relevance. This extensive cataloguing of parameters is instrumental in underpinning the simulation exercises with a high degree of accuracy and relevance, ensuring that the insights gleaned from this study are both valid and applicable to the design. This extensive cataloguing of parameters is instrumental in underpinning the simulation exercises with a high degree of accuracy and relevance, ensuring that the insights gleaned from this study are both valid and applicable to the design and optimization of future smart distribution grids. Through this meticulously constructed model, the paper aims to shed light on the adaptability and control capabilities of the SGAC algorithm, particularly in managing the complexities introduced by the SGAC algorithm. Particularly in managing the complexities introduced by the integration of a diverse mix of renewable energy sources within smart grid infrastructures.</p>
<p>This paper presents a detailed examination of the integration of random white noise as a proxy for load disturbances within a sophisticated smart distribution network model. This model is intentionally crafted to mimic the erratic load fluctuations commonly observed in power systems that are extensively integrated with novel energy resources. The primary objective of this research is to conduct a comprehensive evaluation of the Soft Graph Actor Critic (SGAC)&#x2019;s efficacy and resilience when faced with environments characterized by substantial stochastic disturbances. A key component of this investigation involves conducting simulations that introduce 24-h cycles of random white noise disturbances, thereby allowing for an in-depth analysis of the SGAC algorithm&#x2019;s robustness and long-term operational integrity under scenarios of severe random load fluctuations.</p>
<p>
<xref ref-type="fig" rid="F4">Figure 4</xref> delineates the proficiency of the SGAC algorithm in accurately and promptly tracking random disturbances, thereby highlighting its precision in real-time disturbance management. The outcomes of these simulations, systematically compiled in <xref ref-type="table" rid="T2">Table 2</xref>, offer a quantitative assessment of the generation costs incurred, representing a cumulative analysis of the total regulatory expenses accumulated by all generating units over a 24-h period. A comparative evaluation reveals that the frequency deviation experienced with alternative control algorithms ranges from 1.388 to 3.711 times higher than that encountered when employing the SGAC algorithm. Moreover, the generation costs associated with the SGAC algorithm exhibit a nominal decrease, spanning from 0.0006% to 0.019%. These statistics underscore the SGAC algorithm&#x2019;s superior economic efficiency, advanced self-adaptive capabilities, and its distinguished performance in executing coordinated optimal control compared to other intelligent algorithms.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Data of case 2.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Control algorithms</th>
<th align="center">Average frequency error (Hz)</th>
<th align="center">Generation cost ($)</th>
</tr>
<tr>
<th align="center">&#x7c;&#x394;f <italic>&#x7c;</italic>avg</th>
<th align="center">C<sup>total</sup>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SGAC</td>
<td align="center">0.01736</td>
<td align="center">5494.52</td>
</tr>
<tr>
<td align="center">soft Q-learning</td>
<td align="center">0.01991</td>
<td align="center">5499.85</td>
</tr>
<tr>
<td align="center">PPO</td>
<td align="center">0.02390</td>
<td align="center">5499.98</td>
</tr>
<tr>
<td align="center">TRPO</td>
<td align="center">0.01867</td>
<td align="center">5499.72</td>
</tr>
<tr>
<td align="center">D4PG</td>
<td align="center">0.02084</td>
<td align="center">5499.46</td>
</tr>
<tr>
<td align="center">A3C</td>
<td align="center">0.01901</td>
<td align="center">5499.56</td>
</tr>
<tr>
<td align="center">TD3</td>
<td align="center">0.01954</td>
<td align="center">5499.32</td>
</tr>
<tr>
<td align="center">DDPG</td>
<td align="center">0.02140</td>
<td align="center">5499.22</td>
</tr>
<tr>
<td align="center">DDQN</td>
<td align="center">0.02163</td>
<td align="center">5499.33</td>
</tr>
<tr>
<td align="center">DQN</td>
<td align="center">0.02561</td>
<td align="center">5498.81</td>
</tr>
<tr>
<td align="center">DMPC</td>
<td align="center">0.02351</td>
<td align="center">5498.96</td>
</tr>
<tr>
<td align="center">MPC</td>
<td align="center">0.02188</td>
<td align="center">5499.20</td>
</tr>
<tr>
<td align="center">Fuzzy-FOPI</td>
<td align="center">0.04315</td>
<td align="center">5496.08</td>
</tr>
<tr>
<td align="center">Fuzzy-PI</td>
<td align="center">0.05273</td>
<td align="center">5498.81</td>
</tr>
<tr>
<td align="center">PSO-PI</td>
<td align="center">0.04205</td>
<td align="center">5496.26</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To further substantiate the SGAC algorithm&#x2019;s performance efficacy, an assortment of disturbances, including step waves, square waves, and random waves, were systematically introduced into the system. The resulting data underscore the SGAC&#x2019;s exceptional convergence properties and its elevated learning efficiency, underscoring its unmatched adaptability and robustness within stochastic operational environments. The algorithm&#x2019;s prowess in attenuating random disturbances and enhancing dynamic control effectiveness across interconnected grid landscapes is emphatically validated through these findings.</p>
<p>
<xref ref-type="fig" rid="F4">Figure 4</xref> provides insight into how the output power from various units aligns with load demand across a 24-h cycle, showcasing the system&#x2019;s adeptness at matching load fluctuations and achieving optimal operational states through the synchronized management of multiple energy sources under a cohesive power command strategy. The Energy Storage System (ESS) is identified as a pivotal element, demonstrating its capability to swiftly and accurately modulate power output, thereby contributing significantly to the balance of supply and demand through its flexible charging and discharging functionalities. The real-time optimization conducted by the system controller enables a smoother and more stable regulation process, facilitating rapid and efficient cooperative responses to abrupt load changes within the power system. This, in turn, validates the SGAC algorithm&#x2019;s capacity to support swift and optimal cooperative operations amidst fluctuating system conditions, thereby enhancing the overall operational efficiency and reliability of the power system in managing dynamic and unpredictable environments.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In conclusion, this study presents a comprehensive evaluation of the integration challenges posed by emerging energy sources within electrical grids, specifically highlighting the resultant variability that complicates traditional Load Frequency Control (LFC) mechanisms. Specifically highlighting the resultant variability that complicates traditional LFC mechanisms. The key findings of this research can be summarised as follows. The key findings of this research can be summarised as follows.</p>
<p>Challenge Identification: This study highlights the complexities introduced by renewable energy integration into electrical grids, specifically the variability leading to frequency fluctuations and increased generation costs.</p>
<p>Innovative Approach: Introduction of the Data-Enhanced Optimum Load Frequency Control (DEO-LFC) approach and the Soft Graph Actor Critic (SGAC) algorithm, utilising deep reinforcement learning and graph sequence neural networks for adaptive frequency regulation. Algorithm, utilising deep reinforcement learning and graph sequence neural networks for adaptive frequency regulation.</p>
<p>Methodological Shift: Transition from traditional control mechanisms to agent-based frameworks within DEO-LFC, aiming to enhance grid stability and optimise generation costs amidst high renewable energy penetration.</p>
<p>Validation and Impact: Application of DEO-LFC to the China Southern Grid&#x2019;s isolated island city microgrid model, showcasing its effectiveness in managing grid stability and reducing generation costs in environments with substantial renewable energy sources.</p>
<p>The study underscores the importance of advanced LFC strategies and algorithmic innovations for addressing the challenges of renewable energy integration into electrical grids, offering a pathway towards more stable and cost-efficient grid operations. The study underscores the importance of advanced LFC strategies and algorithmic innovations for addressing the challenges of renewable energy integration into electrical grids, offering a pathway towards more stable and cost-efficient grid operations.</p>
<p>Our future work will enhance the robustness of the algorithm and apply it to the power grid.</p>
</sec>
<sec id="s6">
<title>6 Declaration of conflicting interests</title>
<p>The author(s) declared no potential conflicts of interest with respect to the research, authorship, and/or publication of this article.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>MW: Writing&#x2013;original draft, Writing&#x2013;review and editing. DM: Writing&#x2013;original draft, Writing&#x2013;review and editing. KX: Writing&#x2013;original draft, Writing&#x2013;review and editing. LY: Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was supported by Science and Technology Project of China Southern Power Grid Corporation, under Grant No. GDKJXM20220183.</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>Authors MW, KX, and LY were employed by Dongfang Electronics Corporation. Author DM was employed by Guangzhou Power Supply Bureau of Guangdong Power Grid Co., Ltd.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bengiamin</surname>
<given-names>N. N.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>W. C.</given-names>
</name>
</person-group> (<year>1982</year>). <article-title>Variable structure control of electric power generation</article-title>. <source>IEEE Trans. Power Appar. Syst.</source> <volume>PAS-101</volume>, <fpage>376</fpage>&#x2013;<lpage>380</lpage>. <pub-id pub-id-type="doi">10.1109/TPAS.1982.317117</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y. H.</given-names>
</name>
<name>
<surname>Leitmann</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kai</surname>
<given-names>X. Z.</given-names>
</name>
</person-group> (<year>1991</year>). <article-title>Robust control design for interconnected systems with time-varying uncertainties</article-title>. <source>Int. J. Control</source> <volume>54</volume>, <fpage>1119</fpage>&#x2013;<lpage>1142</lpage>. <pub-id pub-id-type="doi">10.1080/00207179108934201</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Detection of false data injection attacks on load frequency control system with renewable energy based on fuzzy logic and neural networks</article-title>. <source>J. Mod. Power Syst. Clean. Energy</source> <volume>10</volume>, <fpage>1576</fpage>&#x2013;<lpage>1587</lpage>. <pub-id pub-id-type="doi">10.35833/MPCE.2021.000546</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hosseini</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Etemadi</surname>
<given-names>A. H.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Adaptive neuro-fuzzy inference system based automatic generation control</article-title>. <source>Electr. Power Syst. Res.</source> <volume>78</volume>, <fpage>1230</fpage>&#x2013;<lpage>1239</lpage>. <pub-id pub-id-type="doi">10.1016/j.epsr.2007.10.007</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Load frequency control of power system based on improved AFSA-PSO event-triggering scheme</article-title>. <source>Front. Energy Res.</source> <volume>11</volume>, <fpage>1235467</fpage>. <pub-id pub-id-type="doi">10.3389/fenrg.2023.1235467</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>Z. Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Cooperation-based distributed economic MPC for economic load dispatch and load frequency control of interconnected power systems</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>34</volume>, <fpage>3964</fpage>&#x2013;<lpage>3966</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2019.2917632</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Efficient experience replay based deep deterministic policy gradient for AGC dispatch in integrated energy system</article-title>. <source>Appl. Energy</source> <volume>285</volume>, <fpage>116386</fpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2020.116386</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Long</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chong</surname>
<given-names>K. T.</given-names>
</name>
<name>
<surname>Rodr&#xed;guez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guerrero</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Enhancement of frequency regulation in AC microgrid: a fuzzy-MPC controlled virtual synchronous generator</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>12</volume>, <fpage>3138</fpage>&#x2013;<lpage>3149</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2021.3060780</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahboob Ul Hassan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ramli</surname>
<given-names>M. A. M.</given-names>
</name>
<name>
<surname>Milyani</surname>
<given-names>A. H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Robust load frequency control of hybrid solar power systems using optimization techniques</article-title>. <source>Front. Energy Res.</source> <volume>10</volume>, <fpage>902776</fpage>. <pub-id pub-id-type="doi">10.3389/fenrg.2022.902776</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Loh</surname>
<given-names>P. C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>The sliding mode load frequency control for hybrid power system based on disturbance observer</article-title>. <source>Int. J. Electr. Power Energy Syst.</source> <volume>74</volume>, <fpage>446</fpage>&#x2013;<lpage>452</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2015.07.014</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Decentralized sliding mode load frequency control for multi-area power systems</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>28</volume>, <fpage>4301</fpage>&#x2013;<lpage>4309</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2013.2277131</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Coordinated AGC control strategy for an interconnected multi-source power system based on distributed model predictive control algorithm</article-title>. <source>Front. Energy Res.</source> <volume>10</volume>, <fpage>1019464</fpage>. <pub-id pub-id-type="doi">10.3389/fenrg.2022.1019464</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shangguan</surname>
<given-names>X. C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C. K.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Spencer</surname>
<given-names>J. W.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Robust load frequency control for power system considering transmission delay and sampling period</article-title>. <source>IEEE Trans. Ind. Inf.</source> <volume>17</volume>, <fpage>5292</fpage>&#x2013;<lpage>5303</lpage>. <pub-id pub-id-type="doi">10.1109/TII.2020.3026336</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Optimization and H&#x221e; performance analysis for load frequency control of power systems with time-varying delays</article-title>. <source>Front. Energy Res.</source> <volume>9</volume>, <fpage>762480</fpage>. <pub-id pub-id-type="doi">10.3389/fenrg.2021.762480</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Delille</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bayem</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Guillaud</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Francois</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>High wind power penetration in isolated power systems&#x2014;assessment of wind inertial and primary frequency responses</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>28</volume>, <fpage>2412</fpage>&#x2013;<lpage>2420</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2013.2240466</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>Robust load-frequency controller design for power systems</article-title>. <source>IEE Proc. C-Generation, Transm. Distribution</source> <volume>140</volume>, <fpage>11</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1049/ip-c.1993.0003</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>New robust adaptive load-frequency control with system parametric uncertainties</article-title>. <source>IEE Gener. Transm. Dis.</source> <volume>141</volume>, <fpage>184</fpage>&#x2013;<lpage>190</lpage>. <pub-id pub-id-type="doi">10.1049/ip-gtd:19949757</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A novel automatic generation control method based on the ecological population cooperative control for the islanded smart grid</article-title>. <source>Complexity</source> <volume>2018</volume>, <fpage>1</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1155/2018/2456963</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Automatic generation control based on multiple neural networks with actor-critic strategy</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>32</volume>, <fpage>2483</fpage>&#x2013;<lpage>2493</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2020.3006080</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2016a</year>). <article-title>A wolf pack hunting strategy based virtual tribes control for automatic generation control of smart grid</article-title>. <source>Appl. Energy</source> <volume>178</volume>, <fpage>198</fpage>&#x2013;<lpage>211</lpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2016.06.041</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Automatic generation control based on multiple-step greedy attribute and multiple-level allocation strategy</article-title>. <source>CSEE J. Power Energy Syst.</source> <volume>8</volume>, <fpage>281</fpage>&#x2013;<lpage>292</lpage>. <pub-id pub-id-type="doi">10.17775/CSEEJPES.2020.02650</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2016b</year>). <article-title>Wolf pack hunting strategy for automatic generation control of an islanding smart distribution network</article-title>. <source>Energy Convers. manage.</source> <volume>122</volume>, <fpage>10</fpage>&#x2013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.1016/j.enconman.2016.05.039</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Stabilization of load frequency control system via event-triggered intermittent control</article-title>. <source>IEEE T. Circuits-II</source> <volume>69</volume>, <fpage>4934</fpage>&#x2013;<lpage>4938</lpage>. <pub-id pub-id-type="doi">10.1109/TCSII.2022.3197460</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yousef</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Kharusi</surname>
<given-names>K. A.-</given-names>
</name>
<name>
<surname>Albadi</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Hosseinzadeh</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Load frequency control of a multi-area power system: an adaptive fuzzy logic approach</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>29</volume>, <fpage>1822</fpage>&#x2013;<lpage>1830</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2013.2297432</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H. Z.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Multi-agent correlated equilibrium Q(&#x3bb;) learning for coordinated smart generation control of interconnected power grids</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>30</volume>, <fpage>1669</fpage>&#x2013;<lpage>1679</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2014.2357079</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y. M.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>K. W.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Stochastic optimal generation command dispatch based on improved hierarchical reinforcement learning approach</article-title>. <source>IET Gener. Transm. Dis.</source> <volume>5</volume>, <fpage>789</fpage>&#x2013;<lpage>797</lpage>. <pub-id pub-id-type="doi">10.1049/iet-gtd.2010.0600</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q. H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>R (&#x3bb;) imitation learning for automatic generation control of interconnected power grids</article-title>. <source>Automatica</source> <volume>48</volume>, <fpage>2130</fpage>&#x2013;<lpage>2136</lpage>. <pub-id pub-id-type="doi">10.1016/j.automatica.2012.05.043</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bamisile</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>An ${H_\infty }$ load frequency control scheme for multi-area power system under cyber-attacks and time-varying delays</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>38</volume>, <fpage>1336</fpage>&#x2013;<lpage>1349</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2022.3171101</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z. G.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023b</year>). <article-title>Reliable event-triggered load frequency control of uncertain multiarea power systems with actuator failures</article-title>. <source>IEEE Trans. Autom. Sci. Eng.</source> <volume>20</volume>, <fpage>2516</fpage>&#x2013;<lpage>2526</lpage>. <pub-id pub-id-type="doi">10.1109/TASE.2022.3205176</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Adaptive distributed auction-based algorithm for optimal mileage based AGC dispatch with high participation of renewable energy</article-title>. <source>Int. J. Electr. Power Energy Syst.</source> <volume>124</volume>, <fpage>106371</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2020.106371</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Optimal mileage based AGC dispatch of a GenCo</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>35</volume>, <fpage>2516</fpage>&#x2013;<lpage>2526</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2020.2966509</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>