<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Energy Res.</journal-id>
<journal-title>Frontiers in Energy Research</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Energy Res.</abbrev-journal-title>
<issn pub-type="epub">2296-598X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1339543</article-id>
<article-id pub-id-type="doi">10.3389/fenrg.2023.1339543</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Energy Research</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>RETRACTED: A data-driven approach for generating load profiles based on InfoGAN and MKDE</article-title>
<alt-title alt-title-type="left-running-head">Lan et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenrg.2023.1339543">10.3389/fenrg.2023.1339543</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Lan</surname>
<given-names>Jian</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2571501/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhou</surname>
<given-names>Yanzhen</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/1849522/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Guo</surname>
<given-names>Qinglai</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sun</surname>
<given-names>Hongbin</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff>
<institution>State Key Laboratory of Power System and Generation Equipment</institution>, <institution>Department of Electrical Engineering</institution>, <institution>Tsinghua University</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2419212/overview">Zening Li</ext-link>, Taiyuan University of Technology, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1607679/overview">Yixun Xue</ext-link>, Taiyuan University of Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2579378/overview">Tao Niu</ext-link>, Chongqing University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Qinglai Guo, <email>guoqinglai@mail.tsinghua.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>12</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="eretracted">
<day>12</day>
<month>11</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>11</volume>
<elocation-id>1339543</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>11</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>11</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Lan, Zhou, Guo and Sun.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Lan, Zhou, Guo and Sun</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>High-quality demand-side management requires an abundance of load profiles to support decision-making processes. However, customer energy consumption data often contains sensitive personal information, and service providers face significant challenges in accessing a substantial amount of energy consumption data. To generate a large volume of customer data without compromising privacy, this study introduces a data-driven approach integrating Information Maximizing Generative Adversarial Networks (InfoGAN) with Multivariate Kernel Density Estimation (MKDE) for the generation of load profiles. InfoGAN is firstly trained based on existing customer load profiles, with the <italic>Q</italic> network disentangling the load into feature variables and the generator producing realistic profiles. Subsequently, MKDE is utilized to assess the distribution of these features, enabling the generation of new profiles by sampling new feature variables. The proposed method circumvents the need for intricate sampling or modeling processes and generates realistic data that represents the inherent uncertainties and fluctuations characterizing customers&#x2019; electricity consumption. The generated data could be used as the substitution for real electricity consumption data, thereby facilitating further applications without compromising privacy concerns.</p>
</abstract>
<kwd-group>
<kwd>InfoGAN</kwd>
<kwd>MKDE</kwd>
<kwd>data generation</kwd>
<kwd>privacy</kwd>
<kwd>demand side management</kwd>
</kwd-group>
<counts>
<page-count count="10"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Sustainable Energy Systems</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>With the development of Advanced Metering Infrastructure (AMI) in smart grid, a large amount of fine-grained customers&#x2019; power consumption data is collected by smart meters, leading to a better perception of the demand side for both power utilities and retailers and higher efficiency of all links in power system (<xref ref-type="bibr" rid="B18">Mohassel et al., 2014</xref>). However, these valuable data also carry inherent sensitive information risks, potentially revealing personal habits and lifestyle choices of customers, which poses great threats to customer privacy. Striking a balance between operational efficiency and privacy protection in the smart grid is an ongoing challenge, necessitating methods to model customer energy behavior while ensuring privacy.</p>
<p>The distribution of real-time customers&#x2019; electricity demand can hardly be calculated because of the variation and acute fluctuating aspects between customers (<xref ref-type="bibr" rid="B9">Grandjean et al., 2012</xref>). Though many privacy issues are involved, load curves are still central to demand-side management rather than statistical indicators of consumption data, especially in the electricity market. For distribution network operators, smart meter data can help to realize better low-voltage network modeling and management (<xref ref-type="bibr" rid="B11">Haben et al., 2016</xref>). For electricity retailers, the exact power consumption demand of their customers is vital to their marketing strategies, and high-resolution data guarantees demand forecasting ability, which may result in lower opportunity costs and higher profit (<xref ref-type="bibr" rid="B5">Da Silva et al., 2013</xref>). For aggregators, household load shapes have the potential to enhance the targeting and tailoring of demand response (DR) as well as improve energy reduction recommendations (<xref ref-type="bibr" rid="B16">Kwac et al., 2014</xref>), and when it comes to real-time demand response, potential evaluation load curves are indispensable. With the help of deep learning, smart meter data could also be used for customer characterization, where customers&#x2019; sociodemographic characteristics could be inferred by their load profiles (<xref ref-type="bibr" rid="B22">Wang et al., 2018</xref>). Besides, Non-Intrusive Load Monitoring (NILM) has been used for classification and energy consumption estimation (<xref ref-type="bibr" rid="B7">Gillis et al., 2015</xref>). Besides, energy theft detection (<xref ref-type="bibr" rid="B14">Hu et al., 2019</xref>) and bad data detection (<xref ref-type="bibr" rid="B17">Li et al., 2009</xref>) are also conducted with demand-side energy consumption data. However, though better energy management could be achieved by analyzing the data collected from smart meters, there are also concerns about the abuse of these personal data (<xref ref-type="bibr" rid="B13">Hu and Vasilakos, 2016</xref>), not only affecting the safe operation of the critical infrastructure but also violating customers&#x2019; privacy. Energy consumption data collected from the demand side could expose customers&#x2019; personal activities to anyone with access to these data and result in property damage and other undesirable outcomes.</p>
<p>To address privacy concerns in the provision of customer energy data to other entities in the electricity market, it is advisable to share the processed data instead of raw customer data. However, traditional data privacy protection methods, such as anonymization and adding random noise, have been found to be not always reliable based on existing studies (<xref ref-type="bibr" rid="B2">Armoogum and Bassoo, 2019</xref>). Moreover, the method of Differential Privacy (DP) is often used to conceal user information, but it may introduce excessive noise, particularly for high-dimensional time series data, which may compromise the utility of the data (<xref ref-type="bibr" rid="B20">Sangogboye et al., 2018</xref>). Utilizing models to synthesize new data is one of the approaches to address issues related to data insufficiency or data privacy concerns (<xref ref-type="bibr" rid="B1">Arif et al., 2017</xref>). systematically review existing load modeling techniques, but the biases are inherent in these model-based methods due to the assumptions made regarding the load operations. In recent years, data-driven generative models such as generative adversarial networks (GAN) (<xref ref-type="bibr" rid="B8">Goodfellow et al., 2014</xref>) have enabled the modeling of power systems without models. GAN was first introduced to renewable scenario generation in (<xref ref-type="bibr" rid="B4">Chen et al., 2018</xref>), and has been used in load generation (<xref ref-type="bibr" rid="B21">Wang et al., 2021</xref>), reconstruction of high-temporal-resolution PV generation data (<xref ref-type="bibr" rid="B23">Zhang et al., 2021</xref>), etc. Besides, GAN has also been introduced to generating electroencephalographic data (<xref ref-type="bibr" rid="B6">Debie et al., 2020</xref>), spatial-temporal data (<xref ref-type="bibr" rid="B19">Qu et al., 2020</xref>), and sensitive data in IIoT operations (<xref ref-type="bibr" rid="B12">Hindistan and Yetkin, 2023</xref>), etc. which could realize data privacy protection through data generation. Since the output-diversity characteristic of GAN could match the stochastic power consumption of demand side, it could also be used for generate new consumption data and replace real data sharing to avoid privacy leakage.</p>
<p>This paper introduces a novel approach using information maximizing generative adversarial networks (InfoGAN) combined with multivariate kernel density estimation (MKDE) for load profile generation. First, the InfoGAN model is utilized to learn from existing customer load profiles, where the <italic>Q</italic> network has the capability to decouple the load into feature variables, and the generator is capable of producing realistic load profiles. Subsequently, when new customer load profiles are needed, MKDE is employed to evaluate the distribution of existing feature variables, from which new feature variables can be sampled and corresponding load profiles can be generated. Notably, the proposed method is also applicable to inferring potential loads based on limited available usage data and generating load profiles for new customers. The key contribution of this paper can be summarized as follows.<list list-type="simple">
<list-item>
<p>(1) A novel data-driven approach for generating load profiles is proposed. Information maximizing generative adversarial networks are first introduced to generate load profiles, which could achieve accurate modeling of customer energy demands through a data-driven approach. This allows for the rapid generation of extensive required customer load data, providing a robust data foundation for service providers.</p>
</list-item>
<list-item>
<p>(2) The proposed method could extract the intrinsic features of the load profiles, which provides new insights for load modeling. Combined with multivariate kernel density estimation, it enables the generation of any desired type of load profile, which has been validated through case studies.</p>
</list-item>
<list-item>
<p>(3) Based on the proposed method, potential utility curves can be efficiently and accurately generated from a limited sample of customer load data, which provides significant assistance in researching potential customer load demands, offering valuable insights for further studies.</p>
</list-item>
</list>
</p>
<p>The paper is organized as follows: <xref ref-type="sec" rid="s2">Section 2</xref> outlines our proposed method&#x2019;s framework; <xref ref-type="sec" rid="s3">Section 3</xref> details the methodology, including algorithm introduction and model implementation; <xref ref-type="sec" rid="s4">Section 4</xref> describes and evaluates the results; and <xref ref-type="sec" rid="s5">Section 5</xref> concludes the paper.</p>
</sec>
<sec id="s2">
<title>2 Framework</title>
<p>As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, there are four major participants involved in data circulation in the smart grid: customers, power utilities, data platforms, and service providers. Power utilities collect data from customers through AMI, as well as taking suggestions from service providers such as aggregators to achieve demand-side management. Service providers analyze received data and provide suggestions to participants in the ancillary services market, which could be energy-saving advice for customers or efficient scheduling policies for power utilities, as well as getting involved in demand response as aggregators or participating in the electricity market as retailers. As stated before, the original data collected from customers contains quality and privacy issues so that data platform is necessary to work as an information hub that gets real data from power utilities and provides service providers with cleaned and masked data. Moreover, with the popularization of the electricity market, data platforms may also be able to trade data someday since data itself is one of the most valuable assets in the electricity market.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Participants with access to demand-side data in smart grid.</p>
</caption>
<graphic xlink:href="fenrg-11-1339543-g001.tif"/>
</fig>
<p>To address these issues, we proposed a data-driven method for smart meter data generation based on InfoGAN and MKDE, which could capture the features of historical load profiles, and realistic data can be generated through generative models based on samples from feature space. The generated data maintains the characteristics of historical data as well as hiding detailed personal information, which is suitable for data circulation to service providers. The framework of the proposed method is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, encompassing both the model training and data generation phases.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Framework of the proposed method.</p>
</caption>
<graphic xlink:href="fenrg-11-1339543-g002.tif"/>
</fig>
<p>In the model training process, historical customer energy consumption data are used as the training for the generative model. The training process is meticulously designed to enable the generator within InfoGAN to learn the distribution of historical data through adversarial training, and the <italic>Q</italic> network is capable of extracting key features without labeled data. When new data is required, historical data or specified types of customer load profiles can be used as reference samples. The <italic>Q</italic> network is then utilized to obtain reference features from these samples. Subsequently, Multivariate Kernel Density Estimation (MKDE) is used to estimate the probability density functions (PDFs) of features to decipher their distribution in a multi-dimensional space non-parametrically, allowing for the generation of feature variables through sampling. Finally, these features are used as input for the generator, which could generate the required load profiles.</p>
<p>It should be mentioned that the feature variables for generation are sampled from the distribution of existing data, ensuring that the generated profiles do not directly correspond to specific customers, which could also preserve privacy. Moreover, this approach is applicable even with limited historical data. Sampling in the feature space reduces computational complexity and leverages the attributes of historical data, yielding more realistic load profiles.</p>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methodology</title>
<p>In this section, information maximizing generative adversarial networks and multivariate kernel density estimators are introduced to load profile generation, and detailed implementation are described.</p>
<sec id="s3-1">
<title>3.1 Load profile generation based on InfoGAN</title>
<p>Generative Adversarial Networks (GANs) were first proposed as an unsupervised generative model, which has the ability to generate high-quality, realistic data. The aim of GAN is to capture the potential distribution of input data and generate new identically distributed data samples, which also corresponds to some data issues in the power system. There are two deep neural networks known as generative model <italic>G</italic> and discriminative model <italic>D</italic> being trained simultaneously which corresponds to a minimax two-player game. The goal of GAN is to train generative model <italic>G</italic> to capture exactly the real distribution of the input data <italic>x</italic> assisted by constantly optimized discriminative model <italic>D</italic>. The competition in this game pushes both of these two models to improve their performance until Nash equilibria are achieved that the samples generated by <italic>G</italic> can&#x2019;t be distinguished from the original data <italic>x</italic> by <italic>D</italic>. The distribution of the input data <italic>x</italic> is defined as <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and noise variable <italic>z</italic> under a known prior distribution <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>z</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> such as Gaussian distribution is used as the input of <italic>G</italic>. The target of <italic>G</italic> is to present a mapping from prior distribution to the data space denoted as <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The output of <italic>D</italic> denoted as <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a single scalar representing the estimation that <italic>x</italic> comes from <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> rather than the generated distribution <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. As a result, the objective function of training <italic>G</italic> is maximizing <italic>D</italic> (<italic>G</italic>(<italic>z</italic>)), and the object function of training <italic>D</italic> is minimizing <italic>D</italic> (<italic>G(z</italic>)) as well as maximizing <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and the value function of GAN can be written as follows:<disp-formula id="equ1">
<mml:math id="m8">
<mml:mrow>
<mml:munder>
<mml:mi mathvariant="italic">min</mml:mi>
<mml:mi>G</mml:mi>
</mml:munder>
<mml:munder>
<mml:mi mathvariant="italic">max</mml:mi>
<mml:mi>D</mml:mi>
</mml:munder>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>z</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mo>&#x2010;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Despite its effectiveness, the vanilla GAN encounters challenges in learning interpretable and disentangled representations of data, which is critical for understanding and controlling the generative process. To put it into practical usage, a method to extract the features of data is needed. As an innovative iteration of GANs, InfoGAN (<xref ref-type="bibr" rid="B3">Chen et al., 2016</xref>) addresses this limitation by learning to disentangle representations of the data in an unsupervised manner. The core innovation of InfoGAN lies in the introduction of an auxiliary network, the <italic>Q</italic> network, which maximizes the mutual information between the feature variables <italic>c</italic> and the observations <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, effectively inducing the generator to learn meaningful and interpretable representations. Here, feature variable <italic>c</italic> represents the conditional variable, encoding interpretable and meaningful attributes of generated samples, while <italic>z</italic> is the noise variables introducing randomness to ensure diversity in generation.</p>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> illustrates the structure of InfoGAN. Unlike conventional GANs, where the generator only receives a noise variable <italic>z</italic> as input, InfoGAN augments this by incorporating feature variables <italic>c</italic>. These variables are designed to represent distinct and interpretable attributes of the generated samples, and the <italic>Q</italic> network is trained to predict these feature variables. Besides, the <italic>Q</italic> network often shares parameters with the discriminator to enhance training efficiency and model compactness in practice, which leverages the feature-discriminating capabilities of <italic>D</italic>, facilitating more effective inference of the conditional variables. This training encourages the generator to produce outputs where variations in the features correspond to variations in specific, interpretable aspects of the generated data. <italic>Q</italic> (<italic>c</italic>&#x7c;<italic>x</italic>) is computed to approximate the posterior <italic>P</italic> (<italic>c</italic>&#x7c;<italic>x</italic>), and it is proved that mutual information <italic>I</italic> (<italic>c</italic>, <italic>G</italic> (<italic>z,c</italic>)) can be quantified by <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mfenced open="|" close="" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Consequently, the minimax game of InfoGAN can be described as follows, where <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a regularization coefficient that balances the conventional GAN objective with mutual information maximization.<disp-formula id="equ2">
<mml:math id="m12">
<mml:mrow>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mi>G</mml:mi>
</mml:munder>
<mml:munder>
<mml:mi>max</mml:mi>
<mml:mi>D</mml:mi>
</mml:munder>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mtext>InfoGAN</mml:mtext>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2010;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="|" close="" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The structure of InfoGAN.</p>
</caption>
<graphic xlink:href="fenrg-11-1339543-g003.tif"/>
</fig>
<p>In this paper, the training of InfoGAN is described as follows.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>Information Maximizing Generative Adversarial Networks.<list list-type="simple">
<list-item>
<p>
<bold>&#x2022;</bold>
<bold>Hypermeter:</bold> <inline-formula id="inf11">
<mml:math id="m13">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, learning rate. <inline-formula id="inf12">
<mml:math id="m14">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, batch size. <inline-formula id="inf13">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext>critic</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the number of <italic>D</italic> updates per <italic>G</italic> updates.</p>
</list-item>
<list-item>
<p>
<bold>&#x2022;</bold>
<bold>Require:</bold> <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, initial <italic>G</italic> parameters. <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, initial <italic>D</italic> parameters. <inline-formula id="inf16">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, initial <italic>Q</italic> parameters.</p>
</list-item>
<list-item>
<p>1:&#x2003; while <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> has not converged, do</p>
</list-item>
<list-item>
<p>2:&#x2003; &#x2003;for <italic>t</italic> &#x3d; 0, &#x2026; , <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext>critic</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> do</p>
</list-item>
<list-item>
<p>3:&#xa0;&#x2003;&#x2003; Sample <inline-formula id="inf19">
<mml:math id="m21">
<mml:mrow>
<mml:msubsup>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x23; a batch from the training data</p>
</list-item>
<list-item>
<p>4:&#x2003;&#x2003;&#x2003; Sample <inline-formula id="inf20">
<mml:math id="m22">
<mml:mrow>
<mml:msubsup>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>z</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf21">
<mml:math id="m23">
<mml:mrow>
<mml:msubsup>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> &#x23; a batch from latent distribution</p>
</list-item>
<list-item>
<p>5:&#x2003;&#x2003;&#x2003; &#x23; Update discriminator <italic>D</italic>:</p>
</list-item>
<list-item>
<p>6:&#x2003;&#xa0;&#x2003; <inline-formula id="inf22">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:msub>
<mml:mo>&#x2190;</mml:mo>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="script">Q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>7:&#x2003;&#x2003;&#x2003; <inline-formula id="inf23">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#x2190;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2219;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>8:&#x2003;&#x2003;&#x2003; &#x23; Update <italic>Q</italic> network:</p>
</list-item>
<list-item>
<p>9:&#x2003;&#x2003;&#x2003; <inline-formula id="inf24">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:msub>
<mml:mo>&#x2190;</mml:mo>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="script">Q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>10:&#x2003;&#xa0;&#x2003; <inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
<mml:mo>&#x2190;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2219;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>11:&#x2003;&#x2003; end for</p>
</list-item>
<list-item>
<p>12:&#x2003;&#x2003; &#x23; Update generator <italic>G</italic>:</p>
</list-item>
<list-item>
<p>13:&#xa0;&#x2003; <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
</mml:msub>
<mml:mo>&#x2190;</mml:mo>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi mathvariant="italic">log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="script">Q</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>14:&#x2003;&#x2003; <inline-formula id="inf27">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
<mml:mo>&#x2190;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2219;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>15:&#x2003; end while</p>
</list-item>
</list>
</p>
</statement>
</p>
<p>In practice, generator <italic>G</italic> and discriminator <italic>D</italic> are both deep neural networks composed of multilayer perceptron, normalization, and leaky rectified linear units (leaky ReLU) with RMSProp algorithm for weight updates. <inline-formula id="inf28">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mtext>critic</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the number of discriminator updates per generator update, balancing their training pace and ensuring the discriminator&#x2019;s effectiveness in guiding the generator&#x2019;s learning process. Besides, 1D convolutional layers are also adopted in the proposed models.</p>
<p>In summary, InfoGAN is introduced to load profile generation, aiming at disentangling the features of power consumption data, which not only maintains the generative strengths of traditional GANs but also significantly improves the model&#x2019;s utility in understanding and manipulating complex load data distributions.</p>
</sec>
<sec id="s3-2">
<title>3.2 Multivariate kernel density estimation for feature modeling</title>
<p>To generate data similar to that of a specific customer, the feature variables corresponding to the customer&#x2019;s historical data can be used as a reference. As multivariate kernel density estimation (MKDE) extends the concept of kernel density estimation (KDE) to multiple dimensions, it could estimate the probability density functions of a vector of variables, enabling the sampling of new feature variables for the generation of new samples.</p>
<p>The typical formula of the MKDE can be expressed as follows, where <inline-formula id="inf29">
<mml:math id="m31">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>f</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the estimated probability density function at point <inline-formula id="inf30">
<mml:math id="m32">
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf31">
<mml:math id="m33">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of data points, <inline-formula id="inf32">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the <italic>i</italic>-th data points, <inline-formula id="inf33">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the kernel function measuring the similarity between the point <inline-formula id="inf34">
<mml:math id="m36">
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and the data point <inline-formula id="inf35">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.<disp-formula id="equ3">
<mml:math id="m38">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>f</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>However, MKDE presents increased computational complexity and challenges in bandwidth selection. The choice of bandwidth, critical in density estimation, becomes more complex as it often requires a matrix to appropriately scale the kernel in each dimension, considering inter-variable correlations. In this paper, we adopted the commonly used Silverman&#x2019;s rule (<xref ref-type="bibr" rid="B24">Zhang et al., 2006</xref>), which suggests the use of a diagonal bandwidth matrix <inline-formula id="inf36">
<mml:math id="m39">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> where each diagonal element is derived from the corresponding univariate bandwidth estimate for each dimension. Each diagonal element can be expressed as follows, where <italic>d</italic> is the dimensionality of the data, <italic>n</italic> is the sample size, and <inline-formula id="inf37">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the standard deviation of the <italic>i</italic>-th dimension.<disp-formula id="equ4">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mn>4</mml:mn>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Besides, the kernel function is pivotal as it determines the manner and extent of smoothing applied to the data in MKDE. Compared to other kernels, the Gaussian kernel facilitates the handling of the tails in the data distribution more effectively. As a result, the Gaussian kernel is employed as kernel function <inline-formula id="inf38">
<mml:math id="m42">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to ensure a continuous, smooth density estimate.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Case study</title>
<p>In this section, the proposed method is trained with historical consumption data, aiming at extracting key features of load profiles and generating realistic load profiles conformed to regular patterns of customers&#x2019; electricity consumption.</p>
<sec id="s4-1">
<title>4.1 Data description</title>
<p>The proposed method was validated on a dataset of small and medium enterprises (SME) in Ireland, which includes electricity consumption data collected every 30&#xa0;min over a period of 1.5&#xa0;years. Notably, the missing or abnormal data in this dataset, which can be the result of faulty data collection instruments, is fully removed instead of extrapolating the missing values to protect the original features. To address potential challenges arising from absolute consumption values, which could negatively impact the training approximation and generalization, all features were normalized through min-max scaling, resulting in a standardized range of [0,1]. This normalization process facilitated the comparison of load profiles across diverse customers and served to enhance the disclosure of dynamic data characteristics. Consequently, 90,155&#xa0;days&#x2019; load profiles from 319 customers were selected, and the data from 256 customers were used to train the generative model, while the rest was used for the testing.</p>
<p>To further understand the characteristics of load profiles, the <italic>K</italic>-means clustering method with <italic>K</italic> &#x3d; 3 was employed to cluster all load data, which also facilitates a more structured comparison within homogenous groups and effectively showcases the proposed method&#x2019;s performance. The clustering results and the consumption profiles of typical customers within each cluster are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. It is observed that cluster 1 primarily exhibits consumption peaks at noon with lower loads in the morning and evening, resembling the electricity usage pattern of commercial office buildings. Cluster 2 demonstrates a gradual increase in load from noon to past midnight, likely representing businesses such as restaurants that operate into the evening. Cluster 3 shows higher loads during the early morning hours, suggesting enterprises that operate at night. The clustering results also reveal significant variability and uncertainty in the load profiles, making them difficult to describe with mathematical models. However, similarities could also be found among load profiles, which suggests that a data-driven approach could be effectively used for modeling.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Clustering results of load profiles.</p>
</caption>
<graphic xlink:href="fenrg-11-1339543-g004.tif"/>
</fig>
</sec>
<sec id="s4-2">
<title>4.2 Load profile generation for existing customers</title>
<p>In this paper, the feature variable <italic>c</italic> is a continuous variable containing ten features with a range of values between 0 and 1, and the dimension of noise variable <italic>z</italic> takes 50 which is sampled from a Gaussian distribution. All weights of <italic>G</italic>, <italic>D,</italic> and <italic>Q</italic> are initialized from a centered Normal distribution with a standard deviation of 0.02, and batch normalization is adopted.</p>
<p>
<xref ref-type="fig" rid="F5">Figure 5</xref> provides a visual representation of the efficacy of the proposed method in capturing and replicating the inherent variability in electricity consumption patterns across different customers. <xref ref-type="fig" rid="F5">Figure 5A</xref> delineates the actual load profiles for customers 7, 64, and 73, aggregating over a period of 100&#xa0;days. These profiles are characterized by distinctive usage patterns, underscoring the individualized nature of electricity consumption. In <xref ref-type="fig" rid="F5">Figure 5B</xref>, the red sections encapsulate the distribution of original feature variables extracted by the <italic>Q</italic> network. The blue section presents the distribution formed by randomly sampling 100 feature vectors after analyzing the original feature distribution with MKDE, illustrating the close approximation to the original feature distribution. <xref ref-type="fig" rid="F5">Figure 5C</xref> showcases the generated load profiles based on the aforementioned sampled features. Remarkably, these synthesized profiles exhibit a high degree of resemblance to the actual load profiles, which is indicative of the model&#x2019;s ability to learn and simulate complex, real-world data distributions. It should be noted that since the generated data is randomly sampled from the feature space, there is no deterministic mapping between the original and the generated load profiles.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Original and Generated Results of three typical customers. <bold>(A)</bold> Original Load Profiles, <bold>(B)</bold> Feature Distiriontions, <bold>(C)</bold> Generated Load Profiles.</p>
</caption>
<graphic xlink:href="fenrg-11-1339543-g005.tif"/>
</fig>
<p>An empirical analysis of the impact of individual feature variables on the generated load profile is also conducted. A day&#x2019;s load data of a customer was chosen as the baseline, and its corresponding features were extracted through the <italic>Q</italic> network. Then, different features were modified separately, which were then used as input of the generator along with a fixed noise variable <italic>z</italic> to produce new load profiles. As shown in <xref ref-type="fig" rid="F6">Figure 6</xref>, the abscissa of the figure corresponds to ten features, while the ordinate reflects the modified value of these features. The baseline load data, represented in orange, serves as a reference against which the impact of feature adjustments can be measured, while the generated load profiles are depicted in blue. Notably, regions with significant deviations from the baseline are encapsulated within red dashed boxes. It can be observed that each feature has a noticeable impact on the load profiles. For example, feature <inline-formula id="inf39">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> predominantly impacts the onset of the morning load peak and a <inline-formula id="inf40">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> appears to determine the peak load value in the afternoon. Besides, the interaction between features is not isolated, as the morning load variation is collectively influenced by features <inline-formula id="inf41">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>6</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> through <inline-formula id="inf42">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>9</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which suggests a complex and interconnected feature space contributes to shaping the load profile. The figure demonstrates the capacity of the proposed method to learn and generate the dynamics of load profiles, which also proves its utility in learning disentangled and interpretable representations of the load profiles.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The impact of feature value on the generated results.</p>
</caption>
<graphic xlink:href="fenrg-11-1339543-g006.tif"/>
</fig>
</sec>
<sec id="s4-3">
<title>4.3 Comparative analysis</title>
<p>In this paper, the two most commonly used generative models, Variational Autoencoders (VAE) (<xref ref-type="bibr" rid="B15">Kingma et al., 2019</xref>) and Vanilla GAN, are used for comparison. Due to the extensive variety of load profiles in the original dataset, evaluating the effectiveness of the proposed method using the complete dataset is challenging. Therefore, a comparative approach using different clusters is adopted. Both VAE and GAN are trained with corresponding data of the test clusters and generate 1000 samples. In the proposed method, MKDE is used to estimate the feature distribution under different clusters and feature variables are then sampled based on these distributions to generate samples.</p>
<p>
<xref ref-type="fig" rid="F7">Figure 7</xref> presents a schematic of the samples generated by different methods compared to the samples of the corresponding clusters. The central line graph in each cluster illustrates the average load profile generated by the respective method, with the surrounding shaded area denoting the standard deviation from this mean, encapsulating the variability and dispersion of the samples. It is apparent that the proposed method closely approximates the original data, as indicated by both mean and standard deviation. The results demonstrate the proposed method maintains good statistical characteristics when generating a large volume of samples.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Distribution by hour of load generated by different methods.</p>
</caption>
<graphic xlink:href="fenrg-11-1339543-g007.tif"/>
</fig>
<p>Furthermore, the Maximum Mean Discrepancy (MMD) metric (<xref ref-type="bibr" rid="B10">Gretton et al., 2012</xref>) is also used to estimate the effectiveness of the proposed method. This metric quantifies the disparity between the distributions of real and generated data, with a smaller MMD value signifying a greater similarity between the two data sets. As shown in <xref ref-type="fig" rid="F8">Figure 8</xref>, the results demonstrate that the data generated by the proposed method exhibit higher similarity across various scenarios compared to traditional generative methods, thereby validating the effectiveness of the proposed approach.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Comparison of MMD of different methods.</p>
</caption>
<graphic xlink:href="fenrg-11-1339543-g008.tif"/>
</fig>
</sec>
<sec id="s4-4">
<title>4.4 Load profile generation for new customers</title>
<p>In demand-side management, there also arises a need to generate additional user load data from a limited dataset to infer probable energy usage scenarios, and the enhanced load data could facilitate comprehensive demand forecasting, load balancing, and tailoring energy efficiency measures. Compared to traditional models like the vanilla Generative Adversarial Networks (GAN), which have uncontrollable output results, the proposed method allows for the generation of customer load profiles that align with known information. Besides, the proposed method does not require prior knowledge of a customer&#x2019;s specific category, as is necessary for models like conditional GANs (cGANs), and leverages existing samples as a reference. When limited real data of a customer is available, the proposed method can also rapidly generate potential load profiles for research and analysis.</p>
<p>Different quantities of samples were selected from the test set as reference samples, and the feature distribution was first obtained using MKDE, followed by generating corresponding samples. As shown in <xref ref-type="fig" rid="F9">Figure 9</xref>, the proposed method is capable of generating similar samples even with as few as ten reference samples. As the quantity of reference samples increases, the Maximum Mean Discrepancy (MMD) between the generated samples and the real samples progressively decreases, which indicates that the generated dataset increasingly resembles the real sample set, demonstrating the method&#x2019;s efficacy in accurately replicating real-world data. Additionally, the diversity of the generated samples also increases with the number of reference samples, suggesting that more reference samples can better depict the feature space corresponding to the real samples.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>The relationship between generated sample quality and reference sample quantity.</p>
</caption>
<graphic xlink:href="fenrg-11-1339543-g009.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>This paper presents a novel methodology combining InfoGAN and MKDE for generating customer load profiles. The approach leverages InfoGAN to learn from existing load data, with its <italic>Q</italic> network effectively disentangling feature variables and the generator producing realistic load profiles. MKDE is then used to assess the distribution of these features for new profile generation. Through this procedure, the privacy of customers is well protected because the real data are separated from third parties by generative models. The case studies have demonstrated the quality of generated samples compared to the real load profiles, which could be proof of the effectiveness of the proposed method. The proposed method provides an effective tool for load data analysis in power systems, offering significant support for the planning and management of power systems.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="http://www.ucd.ie/issda/data/commissionforenergyregulationcer/">http://www.ucd.ie/issda/data/commissionforenergyregulationcer/</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>JL: Writing&#x2013;original draft, Writing&#x2013;review and editing. YZ: Writing&#x2013;original draft, Writing&#x2013;review and editing. QG: Writing&#x2013;review and editing. HS: Writing&#x2013;review and editing.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by National Key R&#x26;D Program of China (2018AAA0101503) and by the Science and Technology Project of State Grid Corporation of China: Fundamental Theory of Human-in-the-loop Hybrid-augmented Intelligence for Power Grid Dispatch and Control.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The authors declare that this study received funding from the Science and Technology Project of State Grid Corporation of China: Fundamental Theory of Human-in-the-loop Hybrid-augmented Intelligence for Power Grid Dispatch and Control. The funder had the following involvement in the study: the decision to submit it for publication.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arif</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mather</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bashualdo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Load modeling&#x2014;a review</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>9</volume>, <fpage>5986</fpage>&#x2013;<lpage>5999</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2017.2700436</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Armoogum</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bassoo</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Privacy of energy consumption data of a household in a smart grid</article-title>,&#x201d; in <source>Smart power distribution systems</source> (<publisher-loc>Amsterdam, Netherlands</publisher-loc>: <publisher-name>Elsevier</publisher-name>), <fpage>163</fpage>&#x2013;<lpage>177</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Houthooft</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Infogan: interpretable representation learning by information maximizing generative adversarial nets</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>29</volume>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2016/hash/7c9d0b1f96aebd7b5eca8c3edaa19ebb-Abstract.html">https://proceedings.neurips.cc/paper_files/paper/2016/hash/7c9d0b1f96aebd7b5eca8c3edaa19ebb-Abstract.html</ext-link>
</comment>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kirschen</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Model-free renewable scenario generation using generative adversarial networks</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>33</volume>, <fpage>3265</fpage>&#x2013;<lpage>3275</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2018.2794541</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Da Silva</surname>
<given-names>P. G.</given-names>
</name>
<name>
<surname>Ili&#x107;</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Karnouskos</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>The impact of smart grid prosumer grouping on forecasting accuracy and its benefits for local electricity market trading</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>5</volume>, <fpage>402</fpage>&#x2013;<lpage>410</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2013.2278868</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Debie</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Moustafa</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Whitty</surname>
<given-names>M. T.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>A privacy-preserving generative adversarial network method for securing EEG brain signals</article-title>,&#x201d; in <conf-name>Proceedings of the 2020 international joint conference on neural networks (IJCNN)</conf-name>, <conf-loc>Glasgow, UK</conf-loc>, <conf-date>July 2020</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1109/IJCNN48605.2020.9206683</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gillis</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Alshareef</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Morsi</surname>
<given-names>W. G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Nonintrusive load monitoring using wavelet design and machine learning</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>7</volume>, <fpage>320</fpage>&#x2013;<lpage>328</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2015.2428706</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goodfellow</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Pouget-Abadie</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mirza</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Warde-Farley</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ozair</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Generative adversarial nets</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>27</volume>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2014/hash/5ca3e9b122f61f8f06494c97b1afccf3-Abstract.html">https://proceedings.neurips.cc/paper_files/paper/2014/hash/5ca3e9b122f61f8f06494c97b1afccf3-Abstract.html</ext-link>
</comment>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grandjean</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Adnot</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Binet</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>A review and an analysis of the residential electric load curve models</article-title>. <source>Renew. Sustain. energy Rev.</source> <volume>16</volume>, <fpage>6539</fpage>&#x2013;<lpage>6565</lpage>. <pub-id pub-id-type="doi">10.1016/j.rser.2012.08.013</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gretton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Borgwardt</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Rasch</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Sch&#xf6;lkopf</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Smola</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>A kernel two-sample test</article-title>. <source>J. Mach. Learn. Res.</source> <volume>13</volume>, <fpage>723</fpage>&#x2013;<lpage>773</lpage>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="http://jmlr.org/papers/v13/gretton12a.html">http://jmlr.org/papers/v13/gretton12a.html</ext-link>
</comment>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haben</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Singleton</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Grindrod</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Analysis and clustering of residential customers energy behavioral demand using smart meter data</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>7</volume>, <fpage>136</fpage>&#x2013;<lpage>144</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2015.2409786</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hindistan</surname>
<given-names>Y. S.</given-names>
</name>
<name>
<surname>Yetkin</surname>
<given-names>E. F.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A hybrid approach with GAN and DP for privacy preservation of IIoT data</article-title>. <source>IEEE Access</source> <volume>11</volume>, <fpage>5837</fpage>&#x2013;<lpage>5849</lpage>. <pub-id pub-id-type="doi">10.1109/access.2023.3235969</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Vasilakos</surname>
<given-names>A. V.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Energy big data analytics and security: challenges and opportunities</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>7</volume>, <fpage>2423</fpage>&#x2013;<lpage>2436</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2016.2563461</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Xi</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Utilizing unlabeled data to detect electricity fraud in AMI: a semisupervised deep learning approach</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>30</volume>, <fpage>3287</fpage>&#x2013;<lpage>3299</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2018.2890663</pub-id>
<pub-id pub-id-type="pmid">30714931</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Welling</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An introduction to variational autoencoders</article-title>. <source>Found. Trends&#xae; Mach. Learn.</source> <volume>12</volume>, <fpage>307</fpage>&#x2013;<lpage>392</lpage>. <pub-id pub-id-type="doi">10.1561/2200000056</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kwac</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Flora</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rajagopal</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Household energy consumption segmentation using hourly data</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>5</volume>, <fpage>420</fpage>&#x2013;<lpage>430</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2013.2278477</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Bowers</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Schnier</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Classification of energy consumption in buildings with outlier detection</article-title>. <source>IEEE Trans. Industrial Electron.</source> <volume>57</volume>, <fpage>3639</fpage>&#x2013;<lpage>3644</lpage>. <pub-id pub-id-type="doi">10.1109/TIE.2009.2027926</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohassel</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Fung</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mohammadi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Raahemifar</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>A survey on advanced metering infrastructure</article-title>. <source>Int. J. Electr. Power and Energy Syst.</source> <volume>63</volume>, <fpage>473</fpage>&#x2013;<lpage>484</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2014.06.025</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Gan-driven personalized spatial-temporal private data sharing in cyber-physical social systems</article-title>. <source>IEEE Trans. Netw. Sci. Eng.</source> <volume>7</volume>, <fpage>2576</fpage>&#x2013;<lpage>2586</lpage>. <pub-id pub-id-type="doi">10.1109/TNSE.2020.3001061</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sangogboye</surname>
<given-names>F. C.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Spanos</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kj&#xe6;rgaard</surname>
<given-names>M. B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A framework for privacy-preserving data publishing with enhanced utility for cyber-physical systems</article-title>. <source>ACM Trans. Sens. Netw. (TOSN)</source> <volume>14</volume>, <fpage>1</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1145/3275520</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Load photo: a novel analysis method for load data</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>12</volume>, <fpage>1394</fpage>&#x2013;<lpage>1404</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2020.3025936</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Gan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kirschen</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep learning-based socio-demographic information identification from smart meter data</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>10</volume>, <fpage>2593</fpage>&#x2013;<lpage>2602</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2018.2805723</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A PV generation data reconstruction method based on improved super-resolution generative adversarial network</article-title>. <source>Int. J. Electr. Power and Energy Syst.</source> <volume>132</volume>, <fpage>107129</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2021.107129</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>King</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Hyndman</surname>
<given-names>R. J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>A Bayesian approach to bandwidth selection for multivariate kernel density estimation</article-title>. <source>Comput. Statistics Data Analysis</source> <volume>50</volume>, <fpage>3009</fpage>&#x2013;<lpage>3031</lpage>. <pub-id pub-id-type="doi">10.1016/j.csda.2005.06.019</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>