<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Energy Res.</journal-id>
<journal-title>Frontiers in Energy Research</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Energy Res.</abbrev-journal-title>
<issn pub-type="epub">2296-598X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1366009</article-id>
<article-id pub-id-type="doi">10.3389/fenrg.2024.1366009</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Energy Research</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Quantum-inspired deep reinforcement learning for adaptive frequency control of low carbon park island microgrid considering renewable energy sources</article-title>
<alt-title alt-title-type="left-running-head">Shen et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenrg.2024.1366009">10.3389/fenrg.2024.1366009</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Shen</surname>
<given-names>Xin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Tang</surname>
<given-names>Jianlin</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2575819/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Pan</surname>
<given-names>Feng</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2320658/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Qian</surname>
<given-names>Bin</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Yitao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2666467/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Measurement Center</institution>, <institution>Yunnan Power Grid Co., Ltd.</institution>, <addr-line>Kunming</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>CSG Electric Power Research institute Co., Ltd.</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Guangdong Provincial Key Laboratory of Intelligent Measurement and Advanced Metering of Power Grid</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Metrology Center of Guangdong Power Grid Co., Ltd.</institution>, <addr-line>Qingyuan</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1384986/overview">Kaiping Qu</ext-link>, China University of Mining and Technology, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2602892/overview">Cheng Yang</ext-link>, Shanghai University of Electric Power, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1312399/overview">Linfei Yin</ext-link>, Guangxi University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2631720/overview">Miao Cheng</ext-link>, The University of Hong Kong, Hong Kong SAR, China, in collaboration with reviewer LY</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Xin Shen, <email>23755803@qq.com</email>&#x200a; Jianlin Tang, <email>tangjl2@csg.cn</email>&#x200a;</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1366009</elocation-id>
<history>
<date date-type="received">
<day>05</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>28</day>
<month>03</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Shen, Tang, Pan, Qian and Zhao.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Shen, Tang, Pan, Qian and Zhao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The low carbon park islanded microgrid faces operational challenges due to the high variability and uncertainty of distributed renewable energy sources. These sources cause severe random disturbances that impair the frequency control performance and increase the regulation cost of the islanded microgrid, jeopardizing its safety and stability. This paper presents a data-driven intelligent load frequency control (DDI-LFC) method to address this problem. The method replaces the conventional LFC controller with an intelligent agent based on a deep reinforcement learning algorithm. To adapt to the complex islanded microgrid environment and achieve adaptive multi-objective optimal frequency control, this paper proposes the quantum-inspired maximum entropy actor-critic (QIS-MEAC) algorithm, which incorporates the quantum-inspired principle and the maximum entropy exploration strategy into the actor-critic algorithm. The algorithm transforms the experience into a quantum state and leverages the quantum features to improve the deep reinforcement learning&#x2019;s experience replay mechanism, enhancing the data efficiency and robustness of the algorithm and thus the quality of DDI-LFC. The validation on the Yongxing Island isolated microgrid model of China Southern Grid (CSG) demonstrates that the proposed method utilizes the frequency regulation potential of distributed generation, and reduces the frequency deviation and generation cost.</p>
</abstract>
<kwd-group>
<kwd>load frequency control</kwd>
<kwd>deep meta-reinforcement learning</kwd>
<kwd>islanded microgrid</kwd>
<kwd>maximum entropy exploration</kwd>
<kwd>quantum-inspired</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Process and Energy Systems Engineering</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Distributed power supply has strong randomness and weak controllability, and its output mode is highly intermittent. Moreover, the load demand-side response is uncertain and the grid interconnection factors are sudden. These all affect the balance of supply and demand and the quality of power in the power system, leading to various problems for industrial and agricultural production and daily life. They cause economic losses and may even endanger the safe operation of the power grid. Frequency is an important measure of power quality. As one of the key indicators of power quality, frequency can directly reflect the balance between the load power on the demand side and the generator&#x2019;s power generation in the power system. Therefore, maintaining the frequency stability is a feasible way to ensure the dynamic stability of the system under strong random disturbances. LFC 1 is a kind of ultra-short-term frequency regulation technology. The LFC controller uses closed-loop feedback control to adjust the output power of the LFC unit according to a certain control strategy. It senses a series of state indicators such as frequency, area control error (ACE), contact line exchange power, and output power of the unit. This achieves the dynamic balance of the power generation and the load power, and then keeps the grid frequency at the specified value and the contact line exchange power at the planned value. Thus, LFC control technology has been widely used in power system operation control. However, the traditional centralised LFC system (<xref ref-type="bibr" rid="B5">Li et al., 2020</xref>; <xref ref-type="bibr" rid="B12">Sun et al., 2023</xref>) always prioritises the optimal control performance of its own region, and the information synergy between regions is low. It is hard to meet the control performance demand of a high proportion of large-capacity new energy grid-connected mode with the traditional centralised AGC as a vital means of grid scheduling. Moreover, the control performance of LFC largely depends on the control strategy 4, while the traditional LFC control strategy 5 is no longer adequate to cope with the regulation and control tasks under the trend of large-scale new energy grid-connectedness and the stochastic fluctuation of uncertain loads on the customer side (<xref ref-type="bibr" rid="B3">Ferrario et al., 2021</xref>; <xref ref-type="bibr" rid="B4">Li et al., 2022</xref>). Therefore, from the perspective of distributed LFC, it is of great significance to seek a class of optimal LFC control strategies for large-scale grid integration of new energy sources based on modern control theory and intelligent optimization methods. These strategies can meet the control performance and operation requirements of power grids under strong stochastic perturbations in the new type of power systems. The traditional methods include two types: the centralised hierarchical LFC strategy and the fully distributed LFC strategy.</p>
<sec id="s1-1">
<title>1.1 Centralized hierarchical LFC strategy</title>
<p>Some notable examples of this strategy include Model Predictive Control (MPC) (<xref ref-type="bibr" rid="B22">Zheng et al., 2012</xref>), Adaptive Control (AC) (<xref ref-type="bibr" rid="B15">Wen et al., 2015</xref>), Learning-Based Control (LBC) (<xref ref-type="bibr" rid="B9">Qadrdan et al., 2017</xref>), and Adaptive Proportional-Integral (PI) Control (<xref ref-type="bibr" rid="B2">El-Fergany and El-Hameed, 2017</xref>). Zheng et al. (<xref ref-type="bibr" rid="B22">Zheng et al., 2012</xref>) introduced a Distributed Model Predictive Control (DMPC) strategy that relies on the mutual coordination of global performance optimization metrics. Wen et al. (<xref ref-type="bibr" rid="B15">Wen et al., 2015</xref>) proposed a Composite Adaptive Centralized Load Frequency Control (CALFC) strategy for regulating the frequency of source-net-load systems, addressing the challenge of source-load cooperative frequency regulation. Qu et al. (<xref ref-type="bibr" rid="B9">Qadrdan et al., 2017</xref>) developed a Data-Driven Centralized Load Frequency Control (DLCFC) method, treating load frequency control as a stochastic dynamic decision-making problem for source-load cooperative frequency regulation. Qadrdan et al. (<xref ref-type="bibr" rid="B2">El-Fergany and El-Hameed, 2017</xref>) designed an LFC method based on the &#x201c;Social Spider&#x201d; Genetic Optimization Algorithm to tackle the tuning of PI parameters in microgrids.</p>
<p>However, these methods do not adequately consider load modeling or the time series dependence of random disturbances from sources like wind power and photovoltaic systems. Furthermore, their impact on the system&#x2019;s frequency control performance is relatively limited.</p>
<p>Centralized LFC control offers the advantage of reflecting the entire network&#x2019;s state, but it also comes with drawbacks. Firstly, the controller and power distributor employ distinct algorithms for control and optimization, resulting in independence and differing objectives, potentially compromising frequency control performance. Secondly, concentrated communication within the microgrid dispatch center can lead to inconsistencies and delays in frequency control due to communication overload, and may even trigger frequency collapse in some instances. Lastly, centralized LFC control makes it challenging to consider the consistent performance of regulation service providers in the performance-based regulation market across different regions, potentially leading to providers prioritizing local units over those in other areas and grid operators.</p>
</sec>
<sec id="s1-2">
<title>1.2 Fully distributed LFC strategy</title>
<p>Research on fully distributed Load Frequency Control (LFC) structures primarily centers on the multi-agent control framework. This framework comprises agent layers that analyze and process received information, determine suitable control strategies, and cooperate with other agent layers to ensure seamless LFC operation. The prevailing methods in this context are multi-agent collaborative consistency and stochastic consistency methods.</p>
<p>Li et al. (<xref ref-type="bibr" rid="B10">Qing et al., 2015</xref>) introduced a Collaborative Consistent Q-Learning (CCQL) algorithm that leverages a distributed power dispatch model to swiftly and optimally dispatch power commands for distributed LFC control, even in scenarios with high communication demands among units. Xi et al. (<xref ref-type="bibr" rid="B17">Xi et al., 2016b</xref>) proposed a Wolf-Pack Hunting Strategy (WPHS) to handle topological changes arising from power constraints. Wang et al. (<xref ref-type="bibr" rid="B14">Wang and Wang, 2019</xref>) devised a discrete-time robust frequency controller for islanded microgrids, capable of achieving frequency restoration and precise active power dispatch through an iterative learning mechanism. Lou et al. (<xref ref-type="bibr" rid="B6">Lou et al., 2020</xref>) aimed to reduce the operational costs of isolated microgrids by considering the active output costs. They implemented a distributed LFC control strategy based on the consistency approach, leading to an optimal LFC strategy that benefits both global and self-reliance aspects through effective communication among various units. This approach facilitates coordination between controllers and distributors, akin to centralized LFC, while ensuring smooth frequency control and minimizing conflicts of interest among different units. However, it relies heavily on communication among units and areas, making it less suitable for multi-area islanded microgrids.</p>
<p>Reinforcement Learning (RL) is a machine learning technique (<xref ref-type="bibr" rid="B21">Yu et al., 2011</xref>; <xref ref-type="bibr" rid="B16">Wiering and Otterlo, 2012</xref>) that operates without precise knowledge of the model. It offers the advantages of self-learning and dynamic stochastic optimization. RL does not rely on predefined systematic knowledge but continually adapts and optimizes strategies by interacting with the environment and learning through trial and error. This allows RL to find optimal solutions for sequential problems. RL-based control algorithms excel in decision-making, self-learning, and self-optimization, primarily due to the relatively straightforward design of reward functions. As the Load Frequency Control (LFC) process follows a Markov Decision Process (MDP), RL based on MDP can enhance LFC control strategies by crafting suitable reward functions to translate contextual information into appropriate control signals. It also aids in selecting control signals for optimal sequential decision-making iterations, improving aspects such as data processing, feature expression, model generalization, intelligence, and sensitivity of the LFC controller.</p>
<p>This paper explores optimal LFC control strategies for new energy grid integration using RL algorithms, focusing on multi-region collaboration and addressing issues arising from the high proportion of large-capacity new energy sources, which introduce strong random disturbances. This approach aims to enhance the compatibility between new energy sources and the power system, ultimately promoting the development of the new power system. RL is a pivotal topic in Artificial Intelligence, with Imthias et al. (<xref ref-type="bibr" rid="B1">Ahamed et al., 2002</xref>) being among the first to apply it to power system LFC. RL is favored for its high control real-time capabilities and robustness, as it responds primarily to the evaluation of the current control effect. It has found extensive use in ensuring the safe and stable control of power systems.</p>
<p>In addition to RL, classical machine learning algorithms have been widely adopted in Automatic Generation Control (AGC) strategies. Yinsha et al. (<xref ref-type="bibr" rid="B20">Yinsha et al., 2019</xref>) introduced a multi-agent RL game model based on MDP, capable of handling single-task multi-decision game problems, which enhances agent intelligence and system robustness. Sause et al. (<xref ref-type="bibr" rid="B11">Sause, 2013</xref>) proposed an algorithm combining Q-learning and SARSA time variance within the collaborative reinforcement learning framework of &#x201c;Next Available Agent,&#x201d; effectively addressing resource competition among multiple agents in a virtual environment. This improves agents&#x2019; exploration abilities in both static and dynamic environments. An algorithm integrating deep deterministic policy gradients and preferred experience replay is presented in (<xref ref-type="bibr" rid="B18">Ye et al., 2019</xref>), rapidly acquiring environmental feedback in a multi-dimensional continuous state-action space. Yin et al. (<xref ref-type="bibr" rid="B19">Yin et al., 2018</xref>) introduced an algorithm based on Double Q Learning (DQL) to mitigate the positive Q bias issue in Q learning algorithms through underestimation of the maximum expected value.</p>
<p>Ensemble learning, a specialized type of machine learning algorithm that enhances decision-making accuracy through collective decision-making, is less commonly applied in AGC. However, Munos et al. (<xref ref-type="bibr" rid="B8">Munos et al., 2016</xref>) introduced an Ensemble Bootstrapping for Q-Learning algorithm, which combines Q-learning within ensemble learning to correct the positive Q-value bias problem in Q-learning algorithms. This algorithm addresses high variance and Q-value deviation in the Q-learning iteration process, achieving effective control.</p>
<p>The methodologies employed for value function estimation in reinforcement learning algorithms are fundamentally divided into two distinct categories, predicated on the alignment between the target policy (the policy under evaluation) and the behavior policy (the policy enacted by the intelligent agent during environmental interaction). These categories are identified as in-policy and off-policy algorithms. In-policy algorithms undertake the evaluation of the target policy through the utilization of sample data directly derived from the target policy itself, a process typically exemplified by the Sarsa algorithm. Conversely, off-policy algorithms engage in the assessment of the target policy via sample data procured from the behavior policy, a method commonly exemplified by the Q-learning algorithm. Within the context of real-world engineering applications, in-policy algorithms may encounter challenges in efficiently generating requisite sample data or may incur elevated operational costs, which can severely restrict their applicability in complex decision-making scenarios. Off-policy algorithms emerge as a solution to these constraints, offering broad utility in practical Load Frequency Control (LFC) engineering projects. Nevertheless, these algorithms are not without their limitations, primarily due to their reduced robustness and the discrepancies in data distribution between the sample data utilized for target policy evaluation and that required for the off-policy algorithm&#x2019;s evaluation process. Such disparities can lead to phenomena known as &#x201c;overestimation&#x201d; or &#x201c;underestimation&#x201d; of action values, which adversely affect the decision-making precision and convergence efficiency of off-policy algorithms. This issue represents a substantial impediment to the broader application of off-policy reinforcement learning algorithms, especially in the domain of frequency control for islanded microgrids.</p>
<p>In the contemporary landscape of science and technology, where interdisciplinary integration is increasingly becoming a norm, the borrowing and application of concepts from the natural world to information processing technologies are gaining momentum. Among these integrations, the incorporation of quantum physics principles into information processing technologies stands out, promising substantial performance improvements. The amalgamation of quantum physics with artificial intelligence algorithms, in particular, has shown to yield significant enhancement effects. The introduction of quantum characteristics into the frameworks of reinforcement learning algorithms, especially within the deep reinforcement learning experience replay mechanism, has attracted considerable academic interest. By integrating quantum features, the robustness of reinforcement learning algorithms can be significantly improved, offering a promising avenue for enhancing algorithmic performance in complex applications such as LFC in islanded microgrids. This innovative approach demonstrates the potential to mitigate the challenges posed by traditional off-policy algorithms, thereby advancing the field of reinforcement learning and its application in critical engineering solutions.</p>
<p>This paper introduces the Quantum-Inspired Maximum Entropy Actor-Critic (QIS-MEAC) algorithm, which incorporates quantum-inspired principles and the maximum entropy exploration strategy into the original actor-critic algorithm. It transforms experiences into a quantum state and utilizes quantum properties to enhance the experience replay mechanism in deep reinforcement learning. Consequently, this enhancement improves the algorithm&#x2019;s data efficiency and robustness, leading to an overall enhancement in the quality of Data-Driven Intelligent Load Frequency Control (DDI-LFC).</p>
<p>Building upon this algorithm, we have developed a Data-Driven Intelligent Load Frequency Control (DDI-LFC) method. This method replaces the conventional LFC controller with an intelligent agent based on a deep reinforcement learning algorithm. This agent is capable of handling the complex environment of isolated island microgrids and achieving adaptive multi-objective optimal frequency control.</p>
<p>Verification using the South Grid Yongxing Island isolated island microgrid model demonstrates the effectiveness of the proposed method. It fully leverages the frequency regulation capabilities of distributed power sources and energy storage, resulting in minimized frequency deviation and generation costs.</p>
<p>The innovations in this paper can be summarized as follows:<list list-type="simple">
<list-item>
<p>1) This paper introduces a novel approach known as Data-Driven Intelligent Load Frequency Control (DDI-LFC) to tackle the problem at hand. Instead of the traditional LFC controller, this method employs an intelligent agent built upon a deep reinforcement learning algorithm.</p>
</list-item>
<list-item>
<p>2) Furthermore, this paper puts forward the Quantum-Inspired Maximum Entropy Actor-Critic (QIS-MEAC) algorithm, which seamlessly integrates quantum-inspired principles and the maximum entropy exploration strategy into the actor-critic algorithm.</p>
</list-item>
</list>
</p>
<p>
<xref ref-type="sec" rid="s2">Section 2</xref> provides an in-depth description of the islanded microgrid system model. In <xref ref-type="sec" rid="s3">Section 3</xref>, we present a novel method, presenting its comprehensive framework. <xref ref-type="sec" rid="s4">Section 4</xref> is dedicated to conducting case studies that assess the effectiveness of the proposed approach. Finally, in <xref ref-type="sec" rid="s5">Section 5</xref>, we conclude the paper by summarizing key insights and discussing the primary research findings.</p>
</sec>
</sec>
<sec id="s2">
<title>2 Model for island microgrid</title>
<sec id="s2-1">
<title>2.1 Microgrids and distributed power sources</title>
<p>An islanded microgrid is a small-scale system that generates and distributes power using various distributed sources, storage devices, converters, loads, and monitoring and protection devices. Microgrids can operate autonomously and independently, with self-control, protection and management functions. The purpose of microgrid is to enable the flexible and efficient use of distributed sources and to address the challenge of connecting a large number and variety of distributed sources to the grid. Microgrid can utilize renewable energy and cogeneration, among other forms of energy, to enhance energy efficiency and power reliability, to lower grid losses and pollution emissions, and to facilitate the transition to smart grid. Photovoltaic, wind, internal combustion engines, fuel cells, and storage devices are some of the common distributed sources in microgrids. A quick and effective control strategy is needed to ensure the safe and stable operation of the microgrid, by maintaining the balance of voltage, frequency and power. The transfer function of an islanded microgrid is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Model for Island microgrid.</p>
</caption>
<graphic xlink:href="fenrg-12-1366009-g001.tif"/>
</fig>
<sec id="s2-1-1">
<title>2.2.1 Photovoltaic systems</title>
<p>To model the electrical behavior and power production of the PV power generation system, the mathematical model incorporates the PV array, the MPPT controller, the DC-DC converter, and other components. The following equations express the mathematical model of the PV array: Details as Eq. <xref ref-type="disp-formula" rid="e1">1</xref>.<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>I</mml:mi>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>I</italic> is the PV array output current, <italic>V</italic> is the PV array output voltage, <italic>I</italic>
<sub>
<italic>ph</italic>
</sub> is the photogenerated current, <italic>I</italic>
<sub>
<italic>S</italic>
</sub> is the reverse saturation current, <italic>q</italic> is the electron charge, <italic>A</italic> is the diode quality factor, <italic>k</italic> is the Boltzmann&#x2019;s constant, <italic>T</italic> is the cell temperature, <italic>R</italic>
<sub>
<italic>S</italic>
</sub> is the series resistor, <italic>R</italic>
<sub>
<italic>p</italic>
</sub> is the parallel resistor.</p>
</sec>
<sec id="s2-1-2">
<title>2.2.2 Wind power systems</title>
<p>The mathematical model of the wind power system includes wind turbine, wind wheel, generator, inverter etc. to simulate the mechanical and electrical characteristics of the wind power system. The mathematical model of the wind turbine can be represented by the following equations. Details as Eq. <xref ref-type="disp-formula" rid="e2">2</xref>.<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>w</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msubsup>
<mml:mi>v</mml:mi>
<mml:mi>w</mml:mi>
<mml:mn>3</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <italic>P</italic>
<sub>
<italic>w</italic>
</sub> is the wind turbine output power, <italic>&#x3c1;</italic> is the air density, <italic>A</italic> is the swept area of the wind turbine, <italic>C</italic>
<sub>
<italic>p</italic>
</sub> is the wind turbine power coefficient, <italic>&#x3bb;</italic> is the wind turbine rotational speed ratio, <italic>&#x3b2;</italic> is the wind turbine blade inclination angle, <italic>v</italic>
<sub>
<italic>w</italic>
</sub> is the wind speed.</p>
</sec>
<sec id="s2-1-3">
<title>2.2.3 Fuel cells</title>
<p>The mathematical model of a fuel cell includes electrochemical reactions, thermodynamics, hydrodynamics, mass transfer, heat transfer, etc. to simulate variables such as voltage, current, temperature, concentration, etc. of the fuel cell. The mathematical model of a fuel cell can be represented by the following equations. Details as Eq. <xref ref-type="disp-formula" rid="e3">3</xref>.<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b7;</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b7;</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:msub>
<mml:mi>&#x3b7;</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>Where <italic>V</italic>
<sub>
<italic>fc</italic>
</sub> is the fuel cell output voltage, <italic>E</italic>
<sub>
<italic>0</italic>
</sub> is the fuel cell open circuit voltage, <italic>&#x3b7;</italic>
<sub>
<italic>a</italic>
</sub> is the anode polarisation loss, <italic>&#x3b7;</italic>
<sub>
<italic>c</italic>
</sub> is the cathode polarization loss and <italic>&#x3b7;</italic>
<sub>
<italic>ohm</italic>
</sub> is the ohmic loss.</p>
</sec>
<sec id="s2-1-4">
<title>2.2.4 Micro gas turbine modelling</title>
<p>Conventional power generators used in microgrids are generally microfuel generators. Compared with diesel generators, these generators have cleaner emissions and lower operation and maintenance costs, so they are mostly used for daily power supply. According to the analysis of (<xref ref-type="bibr" rid="B17">Xi et al., 2016b</xref>), the frequency control model of microfuel generator can be represented by the model in <xref ref-type="fig" rid="F1">Figure 1</xref> Details as Eqs. <xref ref-type="disp-formula" rid="e4">4</xref>, <xref ref-type="disp-formula" rid="e5">5</xref>.<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>O</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>O</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>H</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:msub>
<mml:mi>&#x3b7;</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <italic>C</italic>
<sub>
<italic>MT</italic>
</sub> is the maintenance cost of the power consumption, the value of <italic>C</italic>
<sub>
<italic>MT,fuel</italic>
</sub> is the unit price of MT fuel gas, LHV is the low calorific value of natural gas, and <italic>P</italic>
<sub>
<italic>MT</italic>
</sub> is the operating efficiency of MT.</p>
</sec>
<sec id="s2-1-5">
<title>2.2.5 Diesel generators</title>
<p>Sag control is a technique that enables diesel generators to keep their frequency and voltage output stable. With sag control, each unit can adjust its power output to the voltage sag, without requiring any communication or coordination with other units. With sag control, each unit can adjust its power output to the voltage sag, without requiring any communication or coordination with other units. This enhances the reliability and flexibility of the distributed generation system. Details as Eqs. <xref ref-type="disp-formula" rid="e6">6</xref>, <xref ref-type="disp-formula" rid="e7">7</xref>.<disp-formula id="e6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>G</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>O</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>G</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>O</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>G</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <italic>C</italic>
<sub>
<italic>DG,OM</italic>
</sub> is the cost of the DG, <italic>k</italic>
<sub>
<italic>DG,OM</italic>
</sub> is the DG maintenance factor; <italic>P</italic>
<sub>
<italic>DG</italic>
</sub> is the fuel cost of the DG, and <italic>&#x3b1;</italic>, <italic>&#x3b2;</italic>, and <italic>&#x3b3;</italic> are the fuel cost coefficients.</p>
</sec>
<sec id="s2-1-6">
<title>2.2.6 Electrochemical energy storage devices</title>
<p>Energy storage device: the mathematical model of the energy storage device includes charge/discharge characteristics, energy management system, voltage control, etc. to simulate the charge/discharge process and power output of the energy storage device. The mathematical model of the energy storage device can be represented by the following equations. Details as Eqs. <xref ref-type="disp-formula" rid="e8">8</xref>&#x2013;<xref ref-type="disp-formula" rid="e10">10</xref>.<disp-formula id="e8">
<mml:math id="m8">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="e9">
<mml:math id="m9">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>O</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>max</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mtext>bat</mml:mtext>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>int</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mtext>bat</mml:mtext>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>where <italic>E</italic> is the energy change rate of the energy storage device, <italic>P</italic>
<sub>
<italic>ch</italic>
</sub> is the charging power of the energy storage device, <italic>P</italic>
<sub>
<italic>dis</italic>
</sub> is the discharging power of the energy storage device, SOC is the state of charge of the energy storage device, <italic>E</italic>
<sub>
<italic>max</italic>
</sub> is the maximum energy of the energy storage device, <italic>V</italic>
<sub>
<italic>bat</italic>
</sub> is the output voltage of the energy storage device, <italic>E</italic>
<sub>
<italic>oc</italic>
</sub> is the open-circuit voltage of the energy storage device, <italic>R</italic>
<sub>
<italic>int</italic>
</sub> is the internal resistance of the energy storage device, <italic>I</italic>
<sub>
<italic>bat</italic>
</sub> is the output current of the energy storage device.</p>
</sec>
</sec>
<sec id="s2-2">
<title>2.2 Objective functions and constraints</title>
<p>The traditional LFC method for microgrids only focuses on reducing the frequency error of the isolated microgrid, without taking the cost into account. This paper presents a DD-LFC method that achieves both objectives: minimising the frequency variation and the power generation cost of the units. The DD- LFC method employs an integrated multi-objective optimization, such that the frequency error of the isolated microgrid is reduced to a minimum. LFC method employs an integrated multi-objective optimization, such that the sum of the absolute values of the frequency variation and the power generation cost is minimized. The constraints are shown below. Details as Eqs. <xref ref-type="disp-formula" rid="e11">11</xref>, <xref ref-type="disp-formula" rid="e12">12</xref>.<disp-formula id="e11">
<mml:math id="m11">
<mml:mrow>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
<disp-formula id="e12">
<mml:math id="m12">
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>in</mml:mtext>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>in</mml:mtext>
</mml:msubsup>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>min</mml:mi>
</mml:msubsup>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>in</mml:mtext>
</mml:msubsup>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>max</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mtext>rate</mml:mtext>
</mml:msubsup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>where &#x394;<italic>P</italic>
<sub>order-&#x2211;</sub> is the total command, &#x394;<italic>P</italic>
<sub>
<italic>i</italic>
</sub>
<sup>max</sup> and &#x394;<italic>P</italic>
<sub>
<italic>i</italic>
</sub>
<sup>min</sup> are the limits of the <italic>ith</italic> unit, &#x394;<italic>P</italic>
<sub>
<italic>i</italic>
</sub>
<sup>rate</sup> is the ramp rate of the <italic>ith</italic> unit, and &#x394;<italic>P</italic>
<sub>
<italic>i</italic>
</sub>
<sup>in</sup> is the command of the <italic>ith</italic> unit.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Training for proposed method</title>
<sec id="s3-1">
<title>3.1 MDP modelling of DDI-LFCs</title>
<p>RL aims to determine the optimal policy for a Markov Decision Process (MDP) where an agent engages in continuous exploration. The policy function, denoted as &#x3c0;, maps the state space (S) to the action space (A). The optimal policy is the one that maximizes the cumulative reward.</p>
<p>In the context of microgrid Load Frequency Control (LFC), Markov Decision Process modeling involves the utilization of MDP, a mathematical framework, to characterize and optimize load dispatch and frequency stabilization problems within microgrids. MDP serves as a discrete-time stochastic control process that models decision-making in situations with uncertainty and partial control. It comprises four key components: the state space, action space, state transition probability, and reward function.</p>
<p>The primary objective of modeling using MDP is to identify an optimal strategy for the microgrid. This strategy is essentially a mapping function from the state space to the action space, designed to maximize or minimize the cumulative rewards over the long term for the microgrid. The cumulative reward <italic>G</italic>
<sub>
<italic>t</italic>
</sub> from time <italic>t is</italic> defined as. Details as Eq. <xref ref-type="disp-formula" rid="e13">13</xref>.<disp-formula id="e13">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>n</mml:mi>
</mml:msup>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>where &#x3bb; is the discount factor, which value lower than 1 is typically used to avoid the endless accumulation of expected rewards that causes the learning process to diverge. The distributor employs the PROP allocation method to guarantee the reasonableness of the power distribution for each unit.</p>
<sec id="s3-1-1">
<title>3.1.1 Action space</title>
<p>The agent generates the total command that determines the unit&#x2019;s output. The only variable that the agent can control is its action, which accounts for 10% of this command. The only variable that the agent can control is its action, which accounts for 10% of this command. Details as Eq. <xref ref-type="disp-formula" rid="e14">14</xref>.<disp-formula id="e14">
<mml:math id="m14">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m15">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mtext>order</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the total command.</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 State space</title>
<p>The microgrid system has two state variables: the frequency error and its integral. The frequency error measures the difference between the actual and the target frequency of the microgrid, while the integral accumulates the error over time. The frequency error measures the difference between the actual and the target frequency of the microgrid, while the integral accumulates the error over time. The output variable is the total power generated by the distributed energy sources in the microgrid. Details as Eq. <xref ref-type="disp-formula" rid="e15">15</xref>.<disp-formula id="e15">
<mml:math id="m16">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mi>t</mml:mi>
</mml:msubsup>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>t</mml:mi>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>where <inline-formula id="inf2">
<mml:math id="m17">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the total output.</p>
</sec>
<sec id="s3-1-3">
<title>3.1.3 Reward functions</title>
<p>The controller aims to reduce both the frequency variation and the production cost. To encourage the agent to find the best policy, a penalty for control actions is included in the reward function. The reward function is defined as follows. Details as Eqs. <xref ref-type="disp-formula" rid="e16">16</xref>, <xref ref-type="disp-formula" rid="e17">17</xref>.<disp-formula id="e16">
<mml:math id="m18">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="e17">
<mml:math id="m19">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0.01</mml:mn>
<mml:mi>H</mml:mi>
<mml:mi>Z</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#x394;</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0.01</mml:mn>
<mml:mi>H</mml:mi>
<mml:mi>Z</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>where <italic>r</italic> is the reward and <italic>A</italic> is the punishment function.</p>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Quantum-inspired QIS-MEAC algorithm framework</title>
<sec id="s3-2-1">
<title>3.2.1 QIS-MEAC foundation framework</title>
<p>This paper proposes a novel experience replay mechanism for quantum-inspired deep reinforcement learning algorithms, which leverages some quantum properties and applies them to reinforcement learning. The aim of this improvement is to offer a natural and user-friendly experience replay method that transforms experiences into quantized representations that correspond to their importance and sampling priority, thereby altering their likelihood of being sampled.</p>
<p>Current deep reinforcement learning algorithms still have some room for improvement in terms of data utilization efficiency, reference adjustment complexity, and computational cost, especially as the reinforcement learning application scenarios become more complex and dynamic, making the interaction with the environment very expensive. Therefore, the demand for data utilization efficiency and robustness of the algorithms is also increasing. By incorporating quantum properties into the experience replay mechanism of deep reinforcement learning, we can achieve better results with less effort in practical control tasks. The DDI-LFC method proposed in this paper improves the experience replay mechanism of deep reinforcement learning by using quantum properties, which enables it to effectively learn more samples and prior knowledge, thus enhancing its robustness and allowing the LFC to perform better under various complex load disturbances and achieve multi-objective optimal control.</p>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> above illustrates the experience replay process of the quantum-inspired deep reinforcement learning algorithm, and <xref ref-type="fig" rid="F2">Figure 2</xref> shows its overall structure. In each training iteration cycle, the agent interacts with the environment and reads the required state and reward information at step t, and then generates a state transition et based on its chosen actions. This state transition is first transformed into a quantum state representation, or more precisely, a mathematical expression of the kth qubit in the quantum integrated system, where k is the index of the qubit in the cache pool. Next, the qubit undergoes a quantum preparation operation and becomes a quantum in a superposed state. Then, by observation, the quantum state representation of the experience collapses into either an acceptance or a rejection state, with a probability that reflects its importance, and a small data batch is drawn from the accepted experience and fed into the neural network for training. Moreover, after each training, the extracted experience is returned to the experience pool and converted back into the quantized representation of the experience. This conversion process involves a combination of two kinds of western operations: quantum preparation operation and quantum depreciation operation. The quantum preparation operation adjusts the probability amplitude of the quantized representation of the experience to match its TD-error, and the quantum depreciation operation considers the number of times the experience is replayed, and adding the replay frequency of the experience will diversify the sampled experience, so as to make the experience replay more balanced. The whole process repeats until the algorithm stops, and the following sections will explain the operations in more detail.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Experience pool quantum operations.</p>
</caption>
<graphic xlink:href="fenrg-12-1366009-g002.tif"/>
</fig>
<p>The QIS-MEAC algorithm aims to maximize both the cumulative reward and the entropy. Entropy quantifies the uncertainty of stochastic strategies, and in deep reinforcement learning, higher entropy implies more diverse and exploratory strategies. Therefore, the QIS-MEAC algorithm has a greater ability to explore. The following is the optimal policy function of the QIS-MEAC algorithm with entropy. Details as Eq. <xref ref-type="disp-formula" rid="e18">18</xref>.<disp-formula id="e18">
<mml:math id="m20">
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>argmax</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi mathvariant="script">H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="script">H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:munder>
</mml:mstyle>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>where <inline-formula id="inf3">
<mml:math id="m21">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the optimal policy function, <italic>s</italic>
<sub>
<italic>t</italic>
</sub> denotes the <italic>t</italic> momentary state, <italic>a</italic>
<sub>
<italic>t</italic>
</sub> denotes the <italic>t</italic> momentary action, <inline-formula id="inf4">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the distributional trajectory under the policy <italic>&#x3c0;</italic>, <italic>r</italic> is the reward, <italic>&#x3b3;</italic> denotes the discount factor, <italic>H</italic> denotes the entropy, and <italic>&#x3b1;</italic> is the parameter used to determine the degree of importance of the entropy.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Quantitative representation of experience</title>
<p>In quantum theory, a quantum can be realised by a two-level electron, a rotating system or a photon. For a two-level electron, &#x7c;0&#x3e; can represent the ground state and, in contrast, &#x7c;1&#x3e; the excited state. For a rotating system, &#x7c;0&#x3e; can represent accelerated rotation, while &#x7c;1&#x3e; represents decelerated rotation. For a photon, &#x7c;0&#x3e; is considered as a quantum system, and its two eigenstates &#x7c;0&#x3e; and &#x7c;1&#x3e; represent the acceptance or rejection of the empirical quantum bit, respectively. In order to better demonstrate the empirical quantum bit and its eigenstates, their details are shown in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Experience pool quantum operations.</p>
</caption>
<graphic xlink:href="fenrg-12-1366009-g003.tif"/>
</fig>
<p>Throughout the learning process, the agent continuously tries to interact with the environment, and this learning process can be modelled as a Markov decision process. For each time step <italic>t</italic>, the state of the agent can be written as <italic>s</italic>
<sub>
<italic>t</italic>
</sub>, at which the agent chooses an action <italic>a</italic>
<sub>
<italic>t</italic>
</sub> according to the action strategy and a specific exploration strategy, and after the action, it moves to the next state <italic>s</italic>
<sub>
<italic>t&#x2b;</italic>1</sub>, and obtains a reward <italic>r</italic>
<sub>
<italic>t</italic>
</sub> from the environment. Eventually, the four elements together make up a state transfer, and are put into the experience cache pool after being assigned with the new index <italic>k.</italic> The state transfer process is converted into a state transfer process by converting it into an experience cache. By converting this state transfer process into a quantum representation, we define acceptance and rejection of a state transfer as two eigenstates. The state transfer is then considered as a quantum bit.</p>
<p>Since the quantised expression of the <italic>kth</italic> experience in the experience pool is of the form <inline-formula id="inf5">
<mml:math id="m23">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="|" close="" separators="|">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3e;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> , the state of the experience cache pool consisting of <italic>M</italic> experience quantum bits can be expressed as a tensor product of <italic>M</italic> quantum subsystems of the form. Details as Eq. <xref ref-type="disp-formula" rid="e19">19</xref>.<disp-formula id="e19">
<mml:math id="m24">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="" close="&#x232a;" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c8;</mml:mi>
<mml:mrow>
<mml:mtext>total</mml:mtext>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="" close="&#x232a;" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c8;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2297;</mml:mo>
<mml:mrow>
<mml:mfenced open="" close="&#x232a;" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c8;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2297;</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mrow>
<mml:mfenced open="" close="&#x232a;" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c8;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-2-3">
<title>3.3.3 Replay mechanisms for quantized experiences</title>
<p>The following page shows the pseudo-code for an integrated quantum-inspired deep reinforcement learning algorithm. At each time step, the agent produces a state transition by interacting with the environment. Since a new state transition does not have associated TD-errors, we assign it the TD-error with the highest priority in the experience pool, which means giving it a higher replay priority. This ensures that every new experience will be sampled at least once with the highest priority. This experience is then transformed into a quantum bit. A quantum preparation operation that uses Grover iteration as the fundamental operation is applied to the quantum representation of the experience in the uniform state until it reaches the final state. When the experience pool is full, the state transition is sampled with a probability amplitude that is proportional to the probability amplitude of its quantum representation, and the chosen experiences form a small data batch that is fed into the neural network for training. For those chosen experiences, when they are returned to the experience pool and prepared as uniform states again, their corresponding quantum representations are also subject to a quantum preparation operation to adjust to the new priority of the experience, and a quantum depreciation operation to adapt to the change in the number of times the experience is replayed. This operation is repeated until the algorithm converges.</p>
<p>An experience pool is established in deep reinforcement learning to store the experience data that are utilized to train and adjust the neural network parameters of an agent. The agent interacts with the environment once more under the direction of the neural network with the new parameters after training it with a small amount of data, and simultaneously produces new empirical data. Hence, the data in the experience pool have to be renewed and replaced periodically to attain better training outcomes. For this purpose, the experience pool has a fixed size, and when the pool is full (as shown by k&#x3e;M in the algorithm&#x2019;s pseudo-code) and new experience data are created, the oldest experience is removed to accommodate the new experience (as shown by k reset to 1 in the pseudo-code of the algorithm). Moreover, the neural network parameters are only updated after the experience pool is full, which corresponds to after LF is set to True in the pseudo-code.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Experiment and case studies</title>
<p>This paper validates the proposed algorithm in the LFC model of an isolated island microgrid on Yongxing Island. This refers to a smart energy system consisting of diesel power generation, photovoltaic power generation, and energy storage, built on Yongxing Island, the largest island among the South China Sea islands. This system can be connected to or disconnected from the main power grid as needed. The size and parameters of the microgrid on Yongxing Island are as follows. The microgrid has a total installed capacity of 1.5&#xa0;MW, including 1&#xa0;MW from the diesel generator, 500&#xa0;kW from the photovoltaic power generation, and 200&#xa0;kWh from the energy storage system. The microgrid can achieve 100 per cent priority use of clean energy sources such as photovoltaic, and it can also flexibly access a variety of energy sources in the future, such as wave energy and portable power. The completion of this microgrid increases the power supply capacity of Yongxing Island by eight times, making the power supply stability of the isolated island comparable to that of a city. In this paper, we also perform simulations and tests on the DDI-LFC that employs the QIS-MEAC algorithm and compare it with other control algorithms, such as DDI-LFC based on SQL algorithm (<xref ref-type="bibr" rid="B23">Li et al., 2021</xref>), DDI-LFC based on SAC algorithm (<xref ref-type="bibr" rid="B24">Xi et al., 2016</xref>), DDI-LFC based on PPO algorithm (<xref ref-type="bibr" rid="B17">Xi et al., 2016b</xref>), DDI-LFC based on TRPO algorithm (<xref ref-type="bibr" rid="B25">Xi et al., 2021</xref>), DDI-LFC based on MPC algorithm <xref ref-type="bibr" rid="B25">Xi et al., 2021</xref>), DDI-LFC based on Fuzzy-FOPI algorithm (<xref ref-type="bibr" rid="B25">Xi et al., 2021</xref>), TS-fuzzy-PI (<xref ref-type="bibr" rid="B26">Xi et al., 2022</xref>), PSO-PI (<xref ref-type="bibr" rid="B27">Li and Zhou, 2024</xref>), and GA-PI (<xref ref-type="bibr" rid="B28">Li and Zhou, 2023</xref>). To run the simulation models and methods that we present in this paper, we use a computer with 2 CPUs of 2.10&#xa0;GHz Intel Xeon Platinum processor and 16&#xa0;GB of RAM. The simulation software package that we use is MATALB/Simulink version 9.8.0 (R2020 a).</p>
<sec id="s4-1">
<title>4.1 Case 1: step disturbance</title>
<p>As displayed in <xref ref-type="table" rid="T1">Table 1</xref>, the Quantum-Inspired Maximum Entropy Actor-Critic (QIS-MEAC) algorithm outperforms the other algorithms significantly, resulting in a substantial reduction in frequency deviation ranging from 9.65% to 75.55% and a decrease in generation cost ranging from 0.0004% to 0.012%. The microgrid&#x2019;s frequency response and diesel generator&#x2019;s output power are both affected by various control methods.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Statistical results for Case 1.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Algorithm</th>
<th align="center">Average frequency deviation (Hz)</th>
<th align="center">Power generation costs ($)</th>
</tr>
<tr>
<th align="center">
<italic>&#x7c;&#x394;f</italic> &#x7c;<italic>
<sub>avg</sub>
</italic>
</th>
<th align="center">C<sup>total</sup>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">QIS-MEAC</td>
<td align="center">0.01150</td>
<td align="center">7,253.07</td>
</tr>
<tr>
<td align="center">SQL</td>
<td align="center">0.01261</td>
<td align="center">7,253.88</td>
</tr>
<tr>
<td align="center">SAC</td>
<td align="center">0.01988</td>
<td align="center">7,253.98</td>
</tr>
<tr>
<td align="center">PPO</td>
<td align="center">0.01329</td>
<td align="center">7,253.82</td>
</tr>
<tr>
<td align="center">TRPO</td>
<td align="center">0.01568</td>
<td align="center">7,253.57</td>
</tr>
<tr>
<td align="center">MPC</td>
<td align="center">0.01369</td>
<td align="center">7,253.82</td>
</tr>
<tr>
<td align="center">Fuzzy-FOPI</td>
<td align="center">0.01396</td>
<td align="center">7,253.82</td>
</tr>
<tr>
<td align="center">TS- fuzzy-PI</td>
<td align="center">0.01577</td>
<td align="center">7,253.57</td>
</tr>
<tr>
<td align="center">PSO-PI</td>
<td align="center">0.01655</td>
<td align="center">7,253.48</td>
</tr>
<tr>
<td align="center">GA-PI</td>
<td align="center">0.02019</td>
<td align="center">7,253.10</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The simulation outcomes unequivocally highlight QIS-MEAC as the leading performer among the four intelligent algorithms, with soft Q-learning following closely. This can be attributed to the fact that both QIS-MEAC and soft Q-learning possess the capability of maximum entropy exploration. This enables them to dynamically adjust the learning pace, continuously update the function table through shared experiences, and determine the relative weight of each region. Consequently, each control region can adapt its control strategy effectively, enhancing control flexibility.</p>
<p>Unlike soft Q-learning, QIS-MEAC doesn&#x27;t require averaging strategy evaluations. Instead, it can directly make decisions based on dynamic joint trajectories and historical state-action pairs. Additionally, it exhibits strong adaptability to the learner&#x2019;s instantaneous learning rate, leading to improved coordinated Load Frequency Control (LFC). QIS-MEAC demonstrates remarkable adaptability and superior control performance under varying system operating conditions, thereby confirming the algorithm&#x2019;s effectiveness and scalability.</p>
<p>RL offers advantages over many methods due to its straightforward and universally applicable parameter settings. Nevertheless, the application of RL theory encounters new challenges. Firstly, for large-scale tasks, determining an optimal common exploration goal for the reinforcement learning of multiple individual intelligences becomes complex. Secondly, each intelligence must record the behaviors of other intelligences (leading to reduced stability) to interact with them and attain joint behaviors, consequently slowing down the convergence speed of various methods. In light of these issues, multi-intelligence reinforcement learning techniques with collective characteristics have emerged and gained widespread adoption. The core concern of reinforcement learning is how to solve dynamic tasks in real-time using intelligent entities&#x2019; exploration techniques in dynamic planning and temporal difference methods. The Quantum-Inspired Maximum Entropy Actor-Critic (QIS-MEAC) proposed in this paper is innovative and efficient, thanks to its precise independent self-optimization capabilities.</p>
<p>In <xref ref-type="fig" rid="F4">Figure 4A</xref> below, the illustration demonstrates how the total power output of the unit effectively manages load variations, including scenic and square wave fluctuations. The active output curve of the LFC unit exhibits overshooting to counteract the effects of random power fluctuations. <xref ref-type="fig" rid="F4">Figure 4B</xref> presents the output regulation curves for different LFC unit types. As shown in the figure, when the load increases, smaller hydro and micro-gas units with lower regulation costs are preferred for increasing output. Conversely, when the load decreases, biomass and diesel units with higher regulation costs are prioritized to reduce output, leading to improved frequency control. The LFC output allocation adheres to the equal micro-increment rate principle, ensuring that the final active output of each unit aligns with the economic allocation principle. Other Deep Reinforcement Learning (DRL) algorithms face challenges in producing satisfactory curves due to the lack of performance enhancement techniques. Furthermore, model-based control algorithms encounter difficulties in demonstrating effective control capabilities due to their heavy reliance on models.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Results for case 1. <bold>(A)</bold> Frequency deviation. <bold>(B)</bold> Total regulated output.</p>
</caption>
<graphic xlink:href="fenrg-12-1366009-g004.tif"/>
</fig>
<p>New energy units offer distinct advantages, including rapid start and stop capabilities, high climb rates, and extensive regulation ranges compared to diesel units. They play a pivotal role in the system, taking on most of the output tasks to address power grid load fluctuations. The controller&#x2019;s online optimization results highlight the smoother and more stable regulation process achieved by the proposed method. This ensures that unit outputs quickly stabilize under new operational conditions, enabling optimal collaboration in response to sudden load changes in the power system.</p>
</sec>
<sec id="s4-2">
<title>4.2 Case 2: step disturbance and renewable disturbance</title>
<p>This study presents a smart distribution network model that integrates various new energy sources, including Electric Vehicles (EVs), Wind Power (WP), Small Hydro (SH), Micro-Gas Turbines (MGTs), Fuel Cells (FCs), Solar Power (SP), and Biomass Power (BP). The model is employed to assess the control effectiveness of Quantum-Inspired Maximum Entropy Actor-Critic (QIS-MEAC) in a highly stochastic environment.</p>
<p>Electric vehicles, wind power, and solar power are considered as stochastic load disturbances due to their significant uncertainty in output. Consequently, they are excluded from the Load-Frequency Control (LFC) analysis. The output of the wind turbine is determined by simulating stochastic wind speed, using finite bandwidth white noise as input. The solar power model derives its output from the simulated variations in sunlight intensity throughout the day.</p>
<p>To comprehensively investigate the intricate effects of random load variations within a power system experiencing uncertain large-scale integration of new energy sources, we introduce random white noise load disturbances into the smart distribution network model. Our objective is to evaluate the performance of Quantum-Inspired Maximum Entropy Actor-Critic (QIS-MEAC) under challenging random perturbations.</p>
<p>We utilize 24&#xa0;h of random white noise disturbance as the evaluation criterion to gauge QIS-MEAC&#x2019;s long-term performance in the face of significant random load disturbances. QIS-MEAC demonstrates remarkable accuracy and rapid responsiveness in tracking these random disturbances. The statistical results of the simulation experiments are presented in <xref ref-type="table" rid="T2">Table 2</xref>, where the generation cost represents the total regulation cost of all generating units over 24&#xa0;h.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Data of case 2.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Control algorithms</th>
<th align="center">Average frequency error (Hz)</th>
<th align="center">Generation cost ($)</th>
</tr>
<tr>
<th align="center">
<italic>&#x7c;&#x394;f</italic> &#x7c;<italic>
<sub>avg</sub>
</italic>
</th>
<th align="center">C<sup>total</sup>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">QIS-MEAC</td>
<td align="center">0.029923</td>
<td align="center">18,704.22</td>
</tr>
<tr>
<td align="center">SQL</td>
<td align="center">0.035237</td>
<td align="center">18,719.8</td>
</tr>
<tr>
<td align="center">SAC</td>
<td align="center">0.048217</td>
<td align="center">18,720.18</td>
</tr>
<tr>
<td align="center">PPO</td>
<td align="center">0.033610</td>
<td align="center">18,719.42</td>
</tr>
<tr>
<td align="center">TRPO</td>
<td align="center">0.039404</td>
<td align="center">18,718.66</td>
</tr>
<tr>
<td align="center">MPC</td>
<td align="center">0.034195</td>
<td align="center">18,719.06</td>
</tr>
<tr>
<td align="center">Fuzzy-FOPI</td>
<td align="center">0.035101</td>
<td align="center">18,718.52</td>
</tr>
<tr>
<td align="center">TS- fuzzy-PI</td>
<td align="center">0.040360</td>
<td align="center">18,718.12</td>
</tr>
<tr>
<td align="center">PSO-PI</td>
<td align="center">0.041450</td>
<td align="center">18,718.28</td>
</tr>
<tr>
<td align="center">GA-PI</td>
<td align="center">0.051276</td>
<td align="center">18,716.76</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The distribution network data reveals that the frequency deviation in other algorithms is 1.12&#x2013;1.71 times higher than that in the QIS-MEAC algorithm, while the QIS-MEAC algorithm reduces the generation cost by 0.067%&#x2013;0.085%. Analysis of control performance metrics underscores QIS-MEAC&#x2019;s superior economy, adaptability, coordination, and optimization control performance compared to other intelligent algorithms.</p>
<p>Furthermore, we conducted tests involving various disturbance types, including step waves, square waves, and random waves. The experimental outcomes demonstrate that Multi-Intelligence Actor-Critic exhibits strong convergence performance and high learning efficiency. Notably, in a random environment, it displays exceptional adaptability by effectively suppressing random disturbances and enhancing dynamic control performance in interconnected grid environments. It establishes a balanced relationship between the output power of different unit types and the load demand across a 24-h period. Consequently, it ensures that the total power output of the units accurately tracks load variations, achieving complementary and synergistic optimal operation among multiple energy sources in each time period.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>The manuscript delineates the development and implementation of a Data-Driven Intelligent Load Frequency Control (DDI-LFC) strategy, aimed at facilitating adaptive, multi-objective optimal frequency regulation through the application of a Quantum-Inspired Maximum Entropy Actor-Critic (QIS-MEAC) algorithm. The salient contributions of this research are articulated as follows:<list list-type="simple">
<list-item>
<p>1) Integration Challenges of Distributed Energy Resources: The manuscript identifies the complexity introduced into islanded microgrid operations by the large-scale integration of distributed, renewable energy sources. These sources exhibit high degrees of randomness and intermittency, resulting in severe random perturbations that compromise the frequency control performance and elevate regulation costs, thereby posing significant challenges to the system&#x2019;s safety and stability. In response, the DDI-LFC method is introduced, replacing traditional Load Frequency Control (LFC) mechanisms with a deep reinforcement learning algorithm-based agent, aimed at enhancing frequency regulation amidst these challenges.</p>
</list-item>
<list-item>
<p>2) Quantum-Inspired Algorithmic Enhancement: To navigate the intricate environment of the islanded microgrid and achieve adaptive, multi-objective optimal frequency control, the research proposes the Quantum-Inspired Maximum Entropy Actor-Critic (QIS-MEAC) algorithm. This innovative algorithm integrates quantum-inspired principles and a maximum entropy exploration strategy with the conventional actor-critic algorithm framework. By transforming experiences into quantum states and exploiting quantum properties, the algorithm significantly enhances the efficiency and robustness of data utilization within the deep reinforcement learning experience replay mechanism, thereby augmenting the effectiveness of the DDI-LFC approach.</p>
</list-item>
<list-item>
<p>3) Empirical Validation and Impact: The efficacy of the proposed DDI-LFC method is empirically validated using the Yongxing Island isolated microgrid model within the South China Grid. Results demonstrate the method&#x2019;s proficiency in leveraging the frequency regulation capabilities of distributed power sources and energy storage systems. Consequently, it substantially mitigates frequency deviations and reduces generation costs, underscoring the potential of the DDI-LFC strategy to improve the operational reliability and economic efficiency of islanded microgrids.</p>
</list-item>
</list>
</p>
<p>Through these contributions, the manuscript not only addresses critical challenges associated with the integration of renewable energy sources into microgrids but also showcases the potential of quantum-inspired algorithms in enhancing the landscape of intelligent load frequency control.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>XS: Methodology, Writing&#x2013;original draft, Conceptualization, Formal Analysis, Software. JT: Methodology, Writing&#x2013;original draft, Funding acquisition, Investigation. FP: Formal Analysis, Supervision, Validation, Writing&#x2013;review and editing. BQ: Validation, Visualization, Formal Analysis, Resources, Writing&#x2013;review and editing. YZ: Validation, Visualization, Data curation, Investigation, Writing&#x2013;original draft.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. The authors gratefully acknowledge the support of the China Southern Power Grid Technology Project (YNKJXM20222402).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>Authors XS and YZ were employed by Measurement Center, Yunnan Power Grid Co., Ltd. Authors JT and BQ were employed by CSG Electric Power Research institute Co., Ltd. Author FP was employed by Metrology Center of Guangdong Power Grid Co., Ltd.</p>
<p>The authors declare that this study received funding from China Southern Power Grid. The funder had the following involvement in the study: study design, collection, analysis, interpretation of data, the writing of this article or the decision to submit it for publication.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahamed</surname>
<given-names>T. P. I.</given-names>
</name>
<name>
<surname>Rao</surname>
<given-names>P. S. N.</given-names>
</name>
<name>
<surname>Sastry</surname>
<given-names>P. S.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>A reinforcement learning approach to automatic generation control</article-title>. <source>Electr. Power Syst. Res.</source> <volume>63</volume>, <fpage>9</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1016/s0378-7796(02)00088-3</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>El-Fergany</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>El-Hameed</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Efficient frequency controllers for autonomous two-area hybrid microgrid system using social-spider optimiser</article-title>. <source>IET Generation, Transm. Distribution</source> <volume>11</volume>, <fpage>637</fpage>&#x2013;<lpage>648</lpage>. <pub-id pub-id-type="doi">10.1049/iet-gtd.2016.0455</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ferrario</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bartolini</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Manzano</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Vivas</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Comodi</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>McPhail</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>A model-based parametric and optimal sizing of a battery/hydrogen storage of a real hybrid microgrid supplying a residential load: towards island operation</article-title>. <source>Adv. Appl. Energy</source> <volume>3</volume>, <fpage>100048</fpage>. <pub-id pub-id-type="doi">10.1016/j.adapen.2021.100048</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Evolutionary multi agent deep meta reinforcement learning method for swarm intelligence energy management of isolated multi area microgrid with internet of things</article-title>. <source>IEEE Internet of Things Journal</source>. <pub-id pub-id-type="doi">10.1109/JIOT.2023.3253693</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Prior Knowledge Incorporated Large-Scale Multiagent Deep Reinforcement Learning for Load Frequency Control of Isolated Microgrid Considering Multi-Structure Coordination</article-title>. <source>IEEE Transactions on Industrial Informatics</source>. <pub-id pub-id-type="doi">10.1109/TII.2023.3316253</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Coordinated load frequency control of multi-area integrated energy system using multi-agent deep reinforcement learning</article-title>. <source>Appl. Energy</source> <volume>306</volume>, <fpage>117900</fpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2021.117900</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Real-time implementation of maximum net power strategy based on sliding mode variable structure control for proton-exchange membrane fuel cell system</article-title>. <source>IEEE Trans. Transp. Electrif.</source> <volume>6</volume>, <fpage>288</fpage>&#x2013;<lpage>297</lpage>. <pub-id pub-id-type="doi">10.1109/TTE.2020.2970835</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Efficient experience replay based deep deterministic policy gradient for AGC dispatch in integrated energy system</article-title>. <source>Appl. Energy</source> <volume>285</volume>, <fpage>116386</fpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2020.116386</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lou</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Distributed secondary voltage control in islanded microgrids with consideration of communication network and time delays</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>11</volume>, <fpage>3702</fpage>&#x2013;<lpage>3715</lpage>. <pub-id pub-id-type="doi">10.1109/tsg.2020.2979503</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahboob</surname>
<given-names>Ul H. S.</given-names>
</name>
<name>
<surname>Ramli</surname>
<given-names>M. A. M.</given-names>
</name>
<name>
<surname>Milyani</surname>
<given-names>A. H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Robust load frequency control of hybrid solar power systems using optimization techniques</article-title>. <source>Front. Energy Res.</source> <volume>10</volume>, <fpage>902776</fpage>. <pub-id pub-id-type="doi">10.3389/fenrg.2022.902776</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Munos</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Stepleton</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Harutyunyan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bellemare</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). &#x201c;<article-title>Safe and efficient off-policy reinforcement learning</article-title>,&#x201d; in <source>Advances in neural information processing systems 29</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Lee</surname>
<given-names>D. D.</given-names>
</name>
<name>
<surname>Sugiyama</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Luxburg</surname>
<given-names>U. V.</given-names>
</name>
<name>
<surname>Guyon</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Garnett</surname>
<given-names>R.</given-names>
</name>
</person-group> (<publisher-loc>Barcelona, Spain</publisher-loc>: <publisher-name>Curran Associates, Inc</publisher-name>), <fpage>1054</fpage>&#x2013;<lpage>1062</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qadrdan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jenkins</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Benefits of demand-side response in combined gas and electricity networks</article-title>. <source>Appl. Energy</source> <volume>192</volume>, <fpage>360</fpage>&#x2013;<lpage>369</lpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2016.10.047</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Qing</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Decentralised reinforcement learning collaborative consensus algorithm for generation dispatch in virtual generation tribe</article-title>,&#x201d; in <conf-name>2015 IEEE Innovative Smart Grid Technologies - Asia (ISGT ASIA). IEEE</conf-name>, <conf-loc>Bangkok, Thailand</conf-loc>, <conf-date>3-6 November 2015</conf-date>, <fpage>1197</fpage>&#x2013;<lpage>1201</lpage>. <pub-id pub-id-type="doi">10.1109/ISGT-Asia.2015.7387139</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sause</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Coordinated reinforcement learning agents in a multi-agent virtual environment</article-title>,&#x201d; in <conf-name>2013 IEEE 13th International Conference on Data Mining Workshops. IEEE</conf-name>, <conf-loc>Dallas, Texas, USA</conf-loc>, <conf-date>7-10 December 2013</conf-date>, <fpage>227</fpage>&#x2013;<lpage>230</lpage>. <pub-id pub-id-type="doi">10.1109/ICDMW.2013.156</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Strategy optimization of emergency frequency control based on new load with time delay characteristics</article-title>. <source>Front. Energy Res.</source> <volume>10</volume>, <fpage>1065405</fpage>. <pub-id pub-id-type="doi">10.3389/fenrg.2022.1065405</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zong</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Periodic event-triggered adaptive tracking control design for nonlinear discrete-time systems via reinforcement learning</article-title>. <source>Neural Netw.</source> <volume>154</volume>, <fpage>43</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2022.06.039</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A practical distributed finite-time control scheme for power system transient stability</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>35</volume>, <fpage>3320</fpage>&#x2013;<lpage>3331</lpage>. <pub-id pub-id-type="doi">10.1109/tpwrs.2019.2904729</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Frequency regulation of source-grid-load systems: a compound control strategy</article-title>. <source>IEEE Trans. Industrial Inf.</source> <volume>12</volume>, <fpage>69</fpage>&#x2013;<lpage>78</lpage>. <pub-id pub-id-type="doi">10.1109/tii.2015.2496309</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wiering</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Otterlo</surname>
<given-names>M. V.</given-names>
</name>
</person-group> (<year>2012</year>). <source>Reinforcement learning: state of the art</source>. <publisher-name>Springer Publishing Company, Incorporated</publisher-name>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2016a</year>). <article-title>A wolf pack hunting strategy based virtual tribes control for automatic generation control of smart grid</article-title>. <source>Appl. Energy</source> <volume>178</volume>, <fpage>198</fpage>&#x2013;<lpage>211</lpage>. <pub-id pub-id-type="doi">10.1016/j.apenergy.2016.06.041</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2016b</year>). <article-title>Wolf pack hunting strategy for automatic generation control of an islanding smart distribution network</article-title>. <source>Energy Convers. Manag.</source> <volume>122</volume>, <fpage>10</fpage>&#x2013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.1016/j.enconman.2016.05.039</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Automatic generation control based on multiple neural networks with actor-critic strategy</article-title>. <source>IEEE Trans. Neural Networks Learn. Syst.</source> <volume>32</volume>, <fpage>2483</fpage>&#x2013;<lpage>2493</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2020.3006080</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Automatic generation control based on multiple-step greedy attribute and multiple-level allocation strategy</article-title>. <source>CSEE J. Power Energy Syst.</source> <volume>8</volume>, <fpage>281</fpage>&#x2013;<lpage>292</lpage>. <pub-id pub-id-type="doi">10.17775/CSEEJPES.2020.02650</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Papadaskalopoulos</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Strbac</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep reinforcement learning for strategic bidding in electricity markets</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>11</volume>, <fpage>1343</fpage>&#x2013;<lpage>1355</lpage>. <pub-id pub-id-type="doi">10.1109/tsg.2019.2936142</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Design of a novel smart generation controller based on deep Q learning for large-scale interconnected power system</article-title>. <source>J. Energy Eng.</source> <volume>144</volume>, <fpage>04018033</fpage>. <pub-id pub-id-type="doi">10.1061/(asce)ey.1943-7897.0000519</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yinsha</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wenyi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhiwen</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Research on PSO-fuzzy algorithm optimised control for multi-area AGC system with DFIG wind turbine</article-title>,&#x201d; in <conf-name>2019 14th IEEE Conference on Industrial Electronics and Applications (ICIEA). IEEE</conf-name>, <conf-loc>Xi&#x27;an, China</conf-loc>, <conf-date>19-21 June 2019</conf-date>, <fpage>877</fpage>&#x2013;<lpage>881</lpage>. <pub-id pub-id-type="doi">10.1109/ICIEA.2019.8834127</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Stochastic optimal CPS relaxed control methodology for interconnected power systems using Q-learning method</article-title>. <source>J. Energy Eng.</source> <volume>137</volume>, <fpage>116</fpage>&#x2013;<lpage>129</lpage>. <pub-id pub-id-type="doi">10.1061/(asce)ey.1943-7897.0000017</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Networked coordination-based distributed model predictive control for large-scale system</article-title>. <source>IEEE Trans. Control Syst. Technol.</source> <volume>21</volume>, <fpage>991</fpage>&#x2013;<lpage>998</lpage>. <pub-id pub-id-type="doi">10.1109/tcst.2012.2196280</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>