<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Energy Res.</journal-id>
<journal-title>Frontiers in Energy Research</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Energy Res.</abbrev-journal-title>
<issn pub-type="epub">2296-598X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1485369</article-id>
<article-id pub-id-type="doi">10.3389/fenrg.2024.1485369</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Energy Research</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A multi-task learning based line parameter identification method for medium-voltage distribution network</article-title>
<alt-title alt-title-type="left-running-head">Jiang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenrg.2024.1485369">10.3389/fenrg.2024.1485369</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Jiang</surname>
<given-names>Xuebao</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhou</surname>
<given-names>Chenbin</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2816202/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Pan</surname>
<given-names>Qi</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Liang</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Bowen</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Yang</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Kang</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Fu</surname>
<given-names>Liudi</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff>
<institution>Suzhou Power Supply Company</institution>, <institution>State Grid Jiangsu Electric Power Co., Ltd.</institution>, <addr-line>Suzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2341258/overview">Yang Yu</ext-link>, Nanjing University of Posts and Telecommunications, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2138438/overview">Hao Chen</ext-link>, Shanghai Maritime University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1408940/overview">Lefeng Cheng</ext-link>, Guangzhou University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Chenbin Zhou, <email>enetwork_sz@163.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>11</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1485369</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>08</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>10</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Jiang, Zhou, Pan, Wang, Wu, Xu, Chen and Fu.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Jiang, Zhou, Pan, Wang, Wu, Xu, Chen and Fu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Accurate line parameters are critical for and dispatch in distribution systems. External operating condition variations affect line parameters, reducing the accuracy of state estimation and power flow calculations. While many methods have been proposed and obtained results rather acceptable, there is room for improvement as they don&#x2019;t fully consider line connections in known topologies. Furthermore, inaccuracies in measurement devices and data acquisition systems can introduce noise and outliers, impacting the reliability of parameter identification. To address these challenges, we propose a line parameter identification method based on Graph Attention Networks and Multi-gate Mixture-of-Experts. The topological structure of the power grid and the capabilities of modern data acquisition equipment are utilized to capture. We also introduce a multi-task learning framework to enable joint training of parameter identification across different branches, thereby enhancing computational efficiency and accuracy. Experiments show that the GAT-MMoE model outperforms traditional methods, with notable improvements in both accuracy and robustness.</p>
</abstract>
<kwd-group>
<kwd>line-parameter identification</kwd>
<kwd>multi-task learning</kwd>
<kwd>mixture of experts</kwd>
<kwd>medium-voltage distribution system</kwd>
<kwd>graph attention network</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Smart Grids</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The rapid development of new power systems has increased the complexity of power grid operations. The integration of distributed power sources and energy storage introduces randomness and volatility, presenting new challenges for the control and operation of distribution networks. Nowadays, power grids are mutating into Smart EEPS with highly integrated cyber systems, physical systems, and social systems. Among ML, RL has strong adaptability; thus, it is applied in many aspects of Smart EEPS, such as stability control, AGC (Automatic Generation Control), VQC (Voltage Quadergy Control), OPFC (Optimal Power Flow Control) and other scenarios (<xref ref-type="bibr" rid="B5">Cheng and Yu., 2019</xref>). Accurate line parameters are crucial for state estimation (SE), event detection, fault analysis, and various calculations within the distribution network (<xref ref-type="bibr" rid="B39">Zhang et al., 2020</xref>; <xref ref-type="bibr" rid="B23">Shi et al., 2020</xref>).</p>
<p>Unlike the transmission network, where line parameters can be derived from physical or empirical formulas based on line length, resistivity, and geometric positioning, the distribution network requires different approaches due to its radial topology and numerous feeder nodes (<xref ref-type="bibr" rid="B30">Wang et al., 2016</xref>; <xref ref-type="bibr" rid="B2">Asprou and Kyriakides, 2018</xref>; <xref ref-type="bibr" rid="B11">Li et al., 2018</xref>). Blueprints and planning documents can provide design parameters, but real parameters often differ due to system upgrades. As a result, traditional transmission line parameter identification (TLPI) methods struggle when applied to distribution networks. The key challenge is linking collected data to the line model. Existing parameter estimation methods can be grouped into two categories: model-driven and data-driven.</p>
<p>In medium voltage distribution networks, the complexity of operations and time-varying loads make it hard to build accurate mathematical models. To obtain more accurate line parameters, real-time line parameter identification can be carried out based on measurement data obtained by on-site measuring devices (<xref ref-type="bibr" rid="B25">Singh et al., 2018</xref>; <xref ref-type="bibr" rid="B37">Yu et al., 2018</xref>; <xref ref-type="bibr" rid="B38">Yu et al., 2019</xref>). Currently, the data used in parameter estimation mainly comes from two types of sensors: Supervisory Control and Data Acquisition (SCADA) systems and Phasor Measurement Units (PMUs). SCADA devices have been widely installed in medium voltage distribution networks, capable of collecting the amplitude of the electrical quantities but unable to obtain the phase data. PMUs can provide synchronized electrical quantities, but their high cost has limited large-scale deployment in distribution networks, failing to meet observability requirements under most conditions. Therefore, domestic and foreign scholars have conducted research on distribution network line parameter identification methods using phase-free data (<xref ref-type="bibr" rid="B34">Xiao et al., 2021</xref>). When PMUs are not available (<xref ref-type="bibr" rid="B24">Shi et al., 2024</xref>), applied a linear regression to estimate line parameters, topology, and phase labels, with nodal angles recovered via non-linear regression.</p>
<p>Since measurement devices characteristics can lead to outliers, it is essential to consider the DLPI problem with outliers and propose a new robust method to improve the accuracy of line parameter identification, especially under conditions involving PMU outliers and discrepancies in the accuracy of the coefficient matrix and observation data matrix. Research methods mainly focus on the least, squares method, residual sensitivity analysis, and regression methods (<xref ref-type="bibr" rid="B40">Zhu and Abur, 2010</xref>; <xref ref-type="bibr" rid="B15">Lin and Abur, 2018</xref>). A new iterative weighted least squares (WLS) method for dealing with line parameter deviations from systematic errors is also proposed, using estimates to calculate the gain matrix and prior knowledge to calculate the covariance matrix Pegoraro and his team focused on the estimation of measurement uncertainty and correction factors of D-PMUs, conducting a series of studies (<xref ref-type="bibr" rid="B19">Pegoraro et al., 2017</xref>; <xref ref-type="bibr" rid="B21">Puddu et al., 2018</xref>; <xref ref-type="bibr" rid="B17">Pegoraro et al., 2019a</xref>; <xref ref-type="bibr" rid="B18">Pegoraro et al., 2019b</xref>; <xref ref-type="bibr" rid="B20">Pegoraro et al., 2022</xref>). However, these methods assume widespread deployment of micro-PMUs, which limits their application. Thus, the performance of the linear regression method is limited by the incomplete configuration of measuring equipment in distribution networks. Meanwhile, as noted in <xref ref-type="bibr" rid="B38">Yu et al. (2019)</xref>, imperfect synchronism and time interval deviations in smart meters may not ensure instant measurements for distributed generations (DGs), flexible loads, and electric vehicles with relatively dynamic behaviors.</p>
<p>Thanks to the development of machine learning and deep learning technologies, data-driven methods are gradually being widely applied to analyze and extract deep insights from data based on partial real-time data. In <xref ref-type="bibr" rid="B13">Li et al. (2022)</xref>, a differential evolution algorithm is employed to identify line parameters, even when many original parameters are missing. Chen and his team find out that the integration of heuristic swarm intelligence search algorithms and AI technologies offers a significant approach to addressing the behavioral decision-making challenges (<xref ref-type="bibr" rid="B42">Cheng, 2020</xref>; <xref ref-type="bibr" rid="B4">Cheng et al., 2021</xref>; <xref ref-type="bibr" rid="B3">Cheng et al., 2022</xref>). Another study (<xref ref-type="bibr" rid="B29">Wang and Yu, 2022</xref>) develops a physics-informed graphical learning algorithm, using stochastic gradient descent to update the three-phase series resistance and reactance (<xref ref-type="bibr" rid="B36">Yang et al., 2022</xref>) proposed an RBFNN-MRO method combining a radial basis function neural network with multi-run optimization, which does not require synchronized phasor measure data as it uses a constant feeder parameter model over a specified short period. Other study <xref ref-type="bibr" rid="B12">Li et al. (2024)</xref>; <xref ref-type="bibr" rid="B35">Yang et al. (2023)</xref> introduced a deep-shallow neural network to approximate power flow equations, employing reinforcement learning to optimize while ensuring maximal physical consistency. To reduce the influence of noise and deviation, different robust methods are used to improve accuracy. <xref ref-type="bibr" rid="B26">Sun et al. (2019)</xref> use convolutional neural networks (CNNs) to classify line impedance values and the results deviate from the original within 10%. Also, recent research in parameter identification focuses on overcoming limitations related to data structure, noise, and accuracy. Graph-based models (MNGAN, MFAGCN) are proposed using attention mechanism to enhance identification with non-Euclidean structures (<xref ref-type="bibr" rid="B33">Xia et al., 2022</xref>; <xref ref-type="bibr" rid="B41">Zou et al., 2024</xref>; <xref ref-type="bibr" rid="B32">Wang et al., 2022</xref>).</p>
<p>Above all, future works in DLPI should integrate physics information with deep learning methodologies. This paper introduces a multi-branch method for identifying line parameters using data from both ends of distribution lines. Addressing current limitations in identifying parameters of branched medium voltage distribution networks with topological constraints, the paper proposes a GAT-MMoE based DLPI method. This approach employs a multi-task neural network incorporating graph convolutional networks to tackle the line parameter identification problem. The graph attention network (GAT) uses an attention mechanism to learn the importance of neighboring nodes in a graph. Unlike traditional methods, where the contribution of neighbors is fixed, GAT dynamically adjusts the influence of each neighboring node based on its relevance to the target node. This leads to more accurate and nuanced feature representation. The MMoE model extracts topology features of the distribution network, with node features derived from graph attention networks and a multi-task learning model employing homoscedastic uncertainty loss.</p>
<p>The rest of this paper is structured as follows: <xref ref-type="sec" rid="s2">Section 2</xref> describes the modeling of the task of multi-branch line parameter identification in the distribution network, including problem formulation and the construction of the graph attention and multi-task modules. <xref ref-type="sec" rid="s3">Section 3</xref> covers the overall framework and workflow of the suggested method. <xref ref-type="sec" rid="s4">Section 4</xref> presents results and discussion, along with dataset description and comparison to alternative machine learning methods. Finally, this paper is summarized in <xref ref-type="sec" rid="s5">Section 5</xref>.</p>
</sec>
<sec id="s2">
<title>2 Problem formulation</title>
<sec id="s2-1">
<title>2.1 Parameter identification task and system construct</title>
<p>Current line parameter identification technologies face several challenges: 1) The low investment and high construction costs of smart devices impede the deployment of PMUs at each bus node, making real-time monitoring of voltage phase angle information difficult. 2) The integration of distributed photovoltaic systems on the user side causes reverse power flow and significant voltage fluctuations, making it difficult to maintain accuracy and robustness in the task of distribution line parameter identification.</p>
<p>The objective of DLPI is to find the mapping between node characteristics and line parameters and to identify the line resistance and reactance of each branch. Given the features of power grid branch, it is achieved using the active power, reactive power, and voltage amplitude provided by measuring equipment in medium-voltage distribution networks. The powerful learning ability of neural networks can be utilized to build a power flow model, mine constraints, and learn historical data to train the model. The power flow calculation using polar form of the nodal power <xref ref-type="disp-formula" rid="e1">Equation 1</xref> is:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>j</mml:mi>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>B</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mfrac>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>B</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the active and reactive power injected into node <italic>i</italic>, respectively. <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the active and reactive power from the power source at node <italic>i</italic>; <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the active and reactive power consumed by the load at node <italic>i</italic>; <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the conductance parameters consisting of g and b between nodes <italic>i</italic> and <italic>j</italic>, respectively. <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the voltage amplitudes of nodes <italic>i</italic> and <italic>j</italic>, respectively; <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the difference in the phase angle of the voltage between nodes <italic>i</italic> and <italic>j</italic>. In the modeling stage, we use power flow equations expressed in polar coordinates to accurately represent the system&#x2019;s behavior. However, during the experimental phase, the results are provided in terms of impedance parameters (resistance R and reactance X) as the original IEEE test case data is given in these terms. To facilitate direct comparison with the IEEE standard data, the node admittance matrix is converted into corresponding impedance values.</p>
<p>Meanwhile, parameter identification can be considered as a multi-task regression problem. Considering that the phase angle difference between the two ends of each branch of the distribution network is tiny, we assume that the phase angle difference of adjacent nodes <italic>i</italic> and node <italic>j</italic> is 0 for easy analysis. Thus, the linear voltage drop equation for line <italic>k</italic> is:<disp-formula id="e2">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf12">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf13">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the line resistance and reactance. <xref ref-type="disp-formula" rid="e2">Equation 2</xref> represents the node connection relationship and line parameters, describing the relationship between node voltage and power. When constructing the system, lines are numbered according to the order of the end nodes of the line. For line <italic>k</italic>, the input characteristics of the distribution network can be expressed as <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>Q</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mi>k</mml:mi>
<mml:mn>6</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, allowing the mapping of <italic>R</italic> and <italic>X</italic> to be determined from the input <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>In traditional line parameter identification, the problem can be generalized as a linear regression problem or a quadratic programming problem. However, due to the fitting properties of linear regression, outliers can have significant effects on the regression, resulting in poor robustness. As the distribution network often encounters noise interference, data missing, or other situations due to the complicated operational conditions and numerous measurement devices, the performance of the linear regression model will deteriorate. Therefore, we select the deep learning method to extract the features of nodes on the premise of obtaining reconstituted measurement data samples. Deep learning techniques offer robust feature extraction capabilities, making them well-suited to handle the complexities and noise inherent in medium-voltage distribution networks.</p>
</sec>
<sec id="s2-2">
<title>2.2 System graph construction</title>
<p>Using network topology as a graph to analyze features allows for a more comprehensive utilization of structural information compared to solely relying on measurement data. Graph data <inline-formula id="inf16">
<mml:math id="m18">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, consisting of a vertex set <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and an edge set <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, being non-Euclidean structured, results in better classification accuracy.</p>
<p>In constructing the general distribution network diagram model, the bus is typically regarded as the node and the connecting line as the edge. However, graph learning focuses on node features. Therefore, for parameter identification tasks, we consider parameters as nodes of the graph and common buses between lines as edges to represent the connection relationships between lines. The feature extraction process is illustrated in the following <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Graph extraction process.</p>
</caption>
<graphic xlink:href="fenrg-12-1485369-g001.tif"/>
</fig>
<p>For undirected graph <inline-formula id="inf19">
<mml:math id="m21">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="script">V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, V is the set of n vertices, <inline-formula id="inf20">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>,<inline-formula id="inf21">
<mml:math id="m23">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the set of edges in the graph, and <inline-formula id="inf22">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf23">
<mml:math id="m25">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the adjacency matrix, representing the topology among the nodes. Take <inline-formula id="inf24">
<mml:math id="m26">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as the input, <inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the eigenvector matrix of the normalized Laplace Matrix of the graph, <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the response function of the eigenvalue. Original standard of GCN is defined as:<disp-formula id="e3">
<mml:math id="m29">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2a;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>U</mml:mi>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:msup>
<mml:mi>U</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>Use Chebyshev polynomials <inline-formula id="inf27">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">&#x39b;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">&#x39b;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to approximate and substitute it into <xref ref-type="disp-formula" rid="e3">Equation 3</xref> to obtain <xref ref-type="disp-formula" rid="e4">Equation 4</xref>, in which <inline-formula id="inf28">
<mml:math id="m31">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>L</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mi>L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e4">
<mml:math id="m32">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2a;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>L</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>Then the convolution process is approximately defined as <xref ref-type="disp-formula" rid="e5">Equation 5</xref> by using first-order Chebyshev polynomial to generate the local convolution kernel:<disp-formula id="e5">
<mml:math id="m33">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2a;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>A</mml:mi>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:msup>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>In this formula, <inline-formula id="inf29">
<mml:math id="m34">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>N</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf30">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>j</mml:mi>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The spectral theory is applied and the output of each layer can be written as <xref ref-type="disp-formula" rid="e6">Equation 6</xref>:<disp-formula id="e6">
<mml:math id="m36">
<mml:mrow>
<mml:msup>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>A</mml:mi>
<mml:msup>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mi>W</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf31">
<mml:math id="m37">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the mapping of each layer, <inline-formula id="inf32">
<mml:math id="m38">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf33">
<mml:math id="m39">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the input feature <inline-formula id="inf34">
<mml:math id="m40">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf35">
<mml:math id="m41">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents the feature matrix of the lth layer of the model. <inline-formula id="inf36">
<mml:math id="m42">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2219;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the activation function, <inline-formula id="inf37">
<mml:math id="m43">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the weight matrix in neural network. Since the model&#x2019;s input is the graph structure data <italic>X</italic>, including the adjacency matrix and corresponding attributes, the graph construction process must be completed before model training to represent the data itself and uncover the association relationships between the data.</p>
</sec>
</sec>
<sec id="s3">
<title>3 GAT-MMoE model design for line parameter identification</title>
<sec id="s3-1">
<title>3.1 Graph attention module design</title>
<p>Graph Attention Network (GAT) combines a graph neural network (GNN) with an attention mechanism, specifically tailored for processing graph-structured data by assigning different attention to neighboring nodes on a graph (<xref ref-type="bibr" rid="B28">Velickovic et al., 2017</xref>). It can reduce the computational cost and make it more scalable than methods that consider all neighbors equally, such as traditional Graph Convolutional Networks (GCNs). This flexibility is particularly important for large, sparse graphs. In this paper, GAT module is applied to transform the feature of each node into an inter-node attention coefficient through the graph attention layer, producing a new feature that allows for monitoring changes in neighboring nodes. Thus, information about each branch line and the distribution of impedance values between adjacent branches can be learned, improving the accuracy of parameter identification. We input <inline-formula id="inf38">
<mml:math id="m44">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and the adjacency matrix into the graph convolution layer to learn the node features and structure.</p>
<p>First, the voltage amplitude of nodes under a single time section is input into the graph attention network to calculate the similarity between each node and its neighbors in the distribution network. For each node <italic>i</italic>, calculate the corresponding coefficient between node <italic>j</italic> and itself:<disp-formula id="e7">
<mml:math id="m45">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mtext>ij</mml:mtext>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msup>
<mml:msub>
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="" separators="|">
<mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msup>
<mml:msub>
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf39">
<mml:math id="m46">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="" separators="|">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is the concatenation operation. <inline-formula id="inf40">
<mml:math id="m47">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the learnable weight matrix for <inline-formula id="inf41">
<mml:math id="m48">
<mml:mrow>
<mml:msup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> layer of the attention mechanism, finally <inline-formula id="inf42">
<mml:math id="m49">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2219;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is used to map the concatenated high-dimensional features to a real number. <inline-formula id="inf43">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> indicates the set of nodes adjacent to node <italic>i</italic>, <inline-formula id="inf44">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf45">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the feature value for node <italic>i</italic> and <italic>j</italic> respectively.</p>
<p>After obtaining the correlation coefficient for all the neighboring nodes of node <italic>i</italic>, the attention coefficient is normalized using softmax:<disp-formula id="e8">
<mml:math id="m53">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>LeakyReLU</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>LeakyReLU</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
<disp-formula id="e9">
<mml:math id="m54">
<mml:mrow>
<mml:mtext>LeakyReLU</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf46">
<mml:math id="m55">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the attention coefficient between the k head attention mechanism node <italic>i</italic> and the adjacent node <italic>j</italic>. According to <xref ref-type="disp-formula" rid="e7">Equations 7</xref>, <xref ref-type="disp-formula" rid="e8">8</xref>, new node features are formed by aggregating information using the attention coefficient matrix <italic>a</italic>.</p>
<p>After the attention weights of all nodes are normalized, the information of nodes is extracted through the graph attention layer. For different features, different attention weights need to be assigned. If only single-layer attention is used, the same attention weights are applied to all attributes of the neighbourhood node, which will weaken the learning ability of the model. The specific calculations in each layer are shown in <xref ref-type="disp-formula" rid="e9">Equations 9</xref>&#x2013;<xref ref-type="disp-formula" rid="e11">11</xref>.</p>
<p>In each attention layer, we use this to weigh the messages of a node&#x2019;s neighbours, which are the neighbour&#x2019;s features multiplied by the same learnable weight matrix <italic>W</italic>. We do this for each attention head and concatenate the result of the heads together:<disp-formula id="e10">
<mml:math id="m56">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:munderover>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:munderover>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
<disp-formula id="e11">
<mml:math id="m57">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
</p>
<p>
<italic>K</italic> is the attention head number and we choose Sigmoid as the activation function. The feature h<sub>i</sub>
<sup>&#x27;</sup>
<inline-formula id="inf47">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, calculated by the multi-head attention mechanism, incorporates the contribution of the features of neighboring node <italic>j</italic> to node <italic>i</italic>, therefore having a stronger ability to express features. The feature information is then input into the multi-task module after being learned by the GAT module to identify branch line parameters. Identification tasks for different branch parameters are input into separate expert networks, with each expert responsible for a specific subspace. Moreover, GAT possesses topological extrapolation capabilities. If the topology of the station area changes due to maintenance or other reasons, adaptive anomaly identification can be performed by inputting the new adjacency matrix after training the original adjacency matrix into GAT.</p>
</sec>
<sec id="s3-2">
<title>3.2 Multi-task module design</title>
<p>In the parameter identification work of distribution networks, multiple branches typically require parameter identification, with each branch having multiple target parameters to identify. The magnitudes of branch resistance and reactance are generally quite different, but their characteristics depend on the same factors. Moreover, branch data are strongly interrelated, and variations in the electrical variables of one transmission branch often affect the measurement data of the entire distribution network. Multi-task learning leverages the correlations between multiple tasks to optimize the performance of multi-parameter identification. In this paper, a multi-task strategy is employed to identify multiple targets simultaneously, thereby reducing computational effort. According to previous works, different branches are identified as independent tasks.</p>
<sec id="s3-2-1">
<title>3.2.1 Model choosing and sharing policy</title>
<p>The Mixture of Experts (MoE) approach was initially developed and explored within the field of artificial neural networks, where experts are typically neural network models used to predict numerical values in regression or class labels in classification. To capture differences among multiple branch line parameter identification tasks, a gating network is added for each task, forming the Multi-gate Mixture of Experts (MMoE) model on the basis of MoE (<xref ref-type="bibr" rid="B22">Shazeer et al., 2017</xref>; <xref ref-type="bibr" rid="B16">Ma et al., 2018</xref>). The MMoE enhances performance by allowing multiple tasks to share a set of expert networks, while also assigning different combinations of those experts to each task. The architecture setup which contains a set of expert networks and gating networks enables better task-specific learning and reduces the risk of overfitting, especially when different tasks are related but still require some specialization. For medium-voltage distribution systems, which face constantly changing conditions like load variations and fault scenarios, traditional methods like FCNs often struggle to adapt in real-time across multiple tasks without significant re-calibration. MMoE excels in these environments by providing a more efficient and adaptive solution, improving both accuracy and reliability in parameter identification across different branches.</p>
<p>Multi-task learning can be divided into two mechanisms: hard parameter sharing, where different tasks share the bottom hidden layer, and soft parameter sharing. Both mechanisms have their advantages and disadvantages. In the hard sharing mechanism, parameter sharing is used for feature extraction and output, reducing the risk of overfitting. However, if task differences are large, the model results become less credible (<xref ref-type="bibr" rid="B10">Jacobs et al., 1991</xref>; <xref ref-type="bibr" rid="B8">Eigen et al., 2013</xref>). The MMoE module represents a soft parameter sharing model, using expert networks as shared substructures for parameter sharing. Each task employs a gating network to learn different combination patterns of the expert networks. Compared to the hard parameter sharing model, MMoE handles task differences more effectively and has demonstrated better performance in practice. The principle is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Soft-sharing mechanism.</p>
</caption>
<graphic xlink:href="fenrg-12-1485369-g002.tif"/>
</fig>
<p>The MMoE model primarily consists of two core components: Gate Net and Experts. The role of Gate Net is to establish a connection between the data and the expert model, determining which expert model should process the input sample. Experts form a relatively independent set of models, each responsible for handling a specific input subspace. First, multiple branch identification tasks are decomposed into several sub-tasks, each corresponding to a network, with an expert model trained in each subnet. We use <xref ref-type="disp-formula" rid="e12">Equations 12</xref>&#x2013;<xref ref-type="disp-formula" rid="e14">14</xref> to represent the model construction.</p>
<p>Let x represent the model input, for task <italic>k</italic>, the MMoE model is formulated as<disp-formula id="e12">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
<disp-formula id="e13">
<mml:math id="m60">
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>where <italic>n</italic> represents the number of tasks, <inline-formula id="inf48">
<mml:math id="m61">
<mml:mrow>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the specific tower network where features are fed up and analyzed. Gate networks <italic>G</italic> assign different weights to each expert, with <inline-formula id="inf49">
<mml:math id="m62">
<mml:mrow>
<mml:msup>
<mml:mi>g</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> being the output of the gate network corresponding to expert network <italic>i</italic> for each subtask <italic>k</italic>. The gating network interprets the predictions made by each expert and aids in deciding which expert to trust for a given input. It takes the input pattern provided to the expert models and outputs the contribution that each expert should have in predicting the input:<disp-formula id="e14">
<mml:math id="m63">
<mml:mrow>
<mml:msup>
<mml:mi>g</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>softmax</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
<inline-formula id="inf50">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a trainable matrix, <inline-formula id="inf51">
<mml:math id="m65">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> . Graphs constructed in <xref ref-type="sec" rid="s2">Section 2</xref> are sparse graphs, thus fit properly as embeddings for sparse features.</p>
<p>For the multi-branch parameter identification task, only highly correlated experts are selected to provide accurate answers. The expert model in this paper is implemented using Multilayer Perceptrons (MLPs). Output results are obtained using pooling methods to achieve a weighted sum prediction based on expert weights. The gated model then receives data elements as input, assigns them to different expert models for inference, and outputs weights representing each expert&#x2019;s contribution to processing the data. The pooling system calculates a weighted sum of the classifier outputs for each class and selects the class with the highest weighted sum. To control overall sparsity, the design and parameter adjustment of the gated network are primarily relied upon when there are many learning tasks. The involvement of more expert models increases the complexity of the calculation. If the gated network activates more expert models in a single selection, model performance improves and sparsity is reduced.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Loss function design</title>
<p>In MTL, label loss is the loss in the calculation of real data labels and network prediction labels for each task. Usually, the label loss is determined by the nature of the learning task and is realized through <xref ref-type="disp-formula" rid="e15">Equation 15</xref> by weighted summation of the loss of different tasks:<disp-formula id="e15">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2a;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
</p>
<p>However, simply using linear weighting of multiple task losses has some significant disadvantages. Therefore, according to <xref ref-type="bibr" rid="B6">Cipolla et al. (2018)</xref> and <xref ref-type="bibr" rid="B9">Fernandez-Delgado et al. (2019)</xref>, considering the distinctive contributions of different tasks to helping the final results, we use homoscedastic uncertainty as a basis for weighting losses to adjust the influence of tasks in the final loss function for optimizing the whole framework. Take two tasks, for example, the log likelihood for this output can then be written as <xref ref-type="disp-formula" rid="e16">Equations 16</xref>, <xref ref-type="disp-formula" rid="e17">17</xref>:<disp-formula id="e16">
<mml:math id="m67">
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x7c;</mml:mo>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>W</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>W</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>W</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="e17">
<mml:math id="m68">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x7c;</mml:mo>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>W</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>W</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
<inline-formula id="inf52">
<mml:math id="m69">
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>W</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the output of a neural network with weights <inline-formula id="inf53">
<mml:math id="m70">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> on input <inline-formula id="inf54">
<mml:math id="m71">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf55">
<mml:math id="m72">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a positive scalar, which is learnt in the training process. Then we can attain joint loss of different tasks through <xref ref-type="disp-formula" rid="e18">Equation 18</xref>
<disp-formula id="e18">
<mml:math id="m73">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mi>W</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>
</p>
<p>The multiple final loss is <xref ref-type="disp-formula" rid="e19">Equation 19</xref>:<disp-formula id="e19">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mi>&#x3c9;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2a;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
</mml:mstyle>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>
</p>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Overall framework of the proposed method</title>
<p>The overall framework of the proposed GAT-MMoE for distribution line parameter identification is depicted in <xref ref-type="fig" rid="F3">Figure 3</xref> As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, a multi-task learning model based on an attention graph is constructed. The input of GAT-MMoE we proposed is feature matrix X and the adjacency matrix A, which means that our input features contain <inline-formula id="inf56">
<mml:math id="m75">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> nodes, each node containing six features. The final output of the whole model can be expressed as <xref ref-type="disp-formula" rid="e20">Equation 20</xref>:<disp-formula id="e20">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(20)</label>
</disp-formula>where <inline-formula id="inf57">
<mml:math id="m77">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the mapping function of the k-th branch. <inline-formula id="inf58">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the j-th expert output.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Overall framework of the proposed GAT-MMoE model.</p>
</caption>
<graphic xlink:href="fenrg-12-1485369-g003.tif"/>
</fig>
<p>To overcome the DLPI problems, the system graph is established according to the description in <xref ref-type="sec" rid="s2">Section 2</xref>, with each node corresponding to a physical bus in the distribution network. The input feature of the distribution system is expressed as <inline-formula id="inf59">
<mml:math id="m79">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The number of nodes is decided by the scale of distribution network, which means that our input features contain n nodes, each node contains the above six features.</p>
<p>The GAT module extracts the characteristics of the system and pays attention to different branch information in different subspaces by using multi-head attention mechanism. Then the different subspaces are concatenated to infuse the learnt information. The features information is then input into the multi-task module for training. The multi-task module is an MMoE-backboned MTL module, which contains multiple expert subnetworks. Gate control units are used to calculate the loss for different tasks during the training process and update the parameters related to each task based on the loss. The model&#x2019;s outputs are the estimated branch line impedance.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Case studies</title>
<p>In this section, the IEEE 14-node distribution network (case 1) and the IEEE 33-node distribution network (case 2) are selected as research objects to verify the effectiveness and robustness of the proposed method. The corresponding topological structure of the system is represented by <xref ref-type="fig" rid="F4">Figures 4</xref>, <xref ref-type="fig" rid="F5">5</xref>. Experiments are performed on a computer with Intel Core i7-8700K @ 3.70 GHz CPU, and NVIDIA GeForce RTX 3060 Ti GPU. It utilizes Python3.10, Pytorch2.0.1 and pandapower2.13.1.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>IEEE-14 distribution system.</p>
</caption>
<graphic xlink:href="fenrg-12-1485369-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>IEEE-33 distribution system.</p>
</caption>
<graphic xlink:href="fenrg-12-1485369-g005.tif"/>
</fig>
<sec id="s4-1">
<title>4.1 Dataset description</title>
<p>The power flow formula of power system is as follows:<disp-formula id="e21">
<mml:math id="m80">
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">P</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">V</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="italic">n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">V</mml:mi>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">G</mml:mi>
<mml:mi mathvariant="italic">ij</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">&#x3b8;</mml:mi>
<mml:mi mathvariant="italic">ij</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">B</mml:mi>
<mml:mi mathvariant="italic">ij</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">&#x3b8;</mml:mi>
<mml:mi mathvariant="italic">ij</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">Q</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">V</mml:mi>
<mml:mi mathvariant="italic">i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="italic">j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="italic">n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">V</mml:mi>
<mml:mi mathvariant="italic">j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="italic">G</mml:mi>
<mml:mi mathvariant="italic">ij</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">&#x3b8;</mml:mi>
<mml:mi mathvariant="italic">ij</mml:mi>
</mml:msub>
<mml:mo>&#x2010;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">B</mml:mi>
<mml:mi mathvariant="italic">ij</mml:mi>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">&#x3b8;</mml:mi>
<mml:mi mathvariant="italic">ij</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(21)</label>
</disp-formula>
</p>
<p>According to <xref ref-type="disp-formula" rid="e21">Equation 21</xref>, Node voltage, active power and reactive power data are simulated by pandapower toolkit (<xref ref-type="bibr" rid="B27">Thurner et al., 2018</xref>). The active power injected by nodes in the load data is sampled by the Latin hypercube sampling method at <inline-formula id="inf60">
<mml:math id="m81">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mn>0.8</mml:mn>
<mml:mi>P</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>1.2</mml:mn>
<mml:mi>P</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, where Ps represents the standard active power of the distribution network. Reactive power data set is generated by active power data set and power factor. The power factor <inline-formula id="inf61">
<mml:math id="m82">
<mml:mrow>
<mml:mi mathvariant="italic">cos</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> satisfies the uniform distribution of parameters <inline-formula id="inf62">
<mml:math id="m83">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0.85</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.95</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>. Reactive power is calculated by power factor through <xref ref-type="disp-formula" rid="e22">Equation 22</xref>. The formula for calculating reactive power injected by nodes is as follows:<disp-formula id="e22">
<mml:math id="m84">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>tan</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>arccos</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3c6;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(22)</label>
</disp-formula>where <inline-formula id="inf63">
<mml:math id="m85">
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mn>0.85</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.95</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf64">
<mml:math id="m86">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf65">
<mml:math id="m87">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are the active power and the reactive power injected into node <italic>i</italic> at time <italic>t</italic>. <inline-formula id="inf66">
<mml:math id="m88">
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the power factor of the system at time <italic>t</italic>.</p>
<p>In this paper, 20 types of radiative network topologies are selected, then we add Gaussian noise with &#x3c3; of 0.01, 0.03, 0.5, 1, 2 to each load level, and sample each noise 6 times. A total of 69,120 sets of samples are obtained and divided, including 70% data as training set, 20% data as test set, 10% data as verification set. The hyperparameters of the model, such as the attention coefficient <italic>&#x3b1;</italic>, and learning rates are determined by 10% of the data set.</p>
</sec>
<sec id="s4-2">
<title>4.2 Evaluation index and baseline model setup</title>
<p>A suite of metrics is employed to manifest the performance of the model proposed in this work. Specific calculation formula as shown in <xref ref-type="disp-formula" rid="e23">Equations 23</xref>, <xref ref-type="disp-formula" rid="e24">24</xref>:</p>
<p>Root mean square error (RMSE):<disp-formula id="e23">
<mml:math id="m89">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
<label>(23)</label>
</disp-formula>
</p>
<p>Mean absolute percentage error (MAPE):<disp-formula id="e24">
<mml:math id="m90">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
<label>(24)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf67">
<mml:math id="m91">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf68">
<mml:math id="m92">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> respectively represent the true value and the predicted value, <italic>K</italic> represents the sampling number.</p>
<p>In order to prove the validity of our proposed model, we adopt the following methods as baselines:<list list-type="simple">
<list-item>
<p>1) LR: By minimizing the sum of squares of errors, the linear regression model is used to provide coefficients that quantify the contribution of each feature to the target variable, but may face the problem of overfitting.</p>
</list-item>
<list-item>
<p>2) SVR: Support vector regression is a machine learning method, which adopts the idea of support vector and the Lagrange multiplier to perform regression analysis on data when doing data fitting.</p>
</list-item>
<list-item>
<p>3) FCN: Fully connected neural network is a type of linear neural network, which inevitably faces the problem of poor precision in dealing with nonlinear data sets and overfitting. Fully connected prediction is accomplished by flattening the input matrix.</p>
</list-item>
</list>
</p>
<p>In our implementation, the GAT module utilizes three attention heads, with a hidden representation dimensionality set to 128 and a dropout rate of 0.3. For optimization, we use a batch size of 128 and a learning rate of 0.005. The MMoE module is trained with the Adam optimizer and the learning rate is grid searched from [0.0001, 0.001, 0.01]. To prevent overfitting in both the expert and gating networks, the dropout rate is 0.2. Different weights of each task are assigned using homoscedastic uncertainty.</p>
</sec>
<sec id="s4-3">
<title>4.3 Results and discussion</title>
<p>According to <xref ref-type="fig" rid="F6">Figure 6</xref>, We can find that the gradient of our proposed model decreases rapidly and converges fast after around the 20th epoch with little change in accuracy, which demonstrates the superiority of our model in deep learning-based algorithms.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Training loss over epochs.</p>
</caption>
<graphic xlink:href="fenrg-12-1485369-g006.tif"/>
</fig>
<p>From <xref ref-type="fig" rid="F7">Figure 7</xref>, it can be noticed that the proposed GAT-MMoE model can effectively identify the branch line parameters. For case 1, the max relative errors lie in branch 7 and 1 respectively for <italic>R</italic> and <italic>X</italic>, reaching 3.83% and 4.35%. <xref ref-type="table" rid="T1">Table 1</xref> shows the parameter identification errors of branch resistance and reactance, the corresponding average errors are 3.84% and 2.67% for <italic>R</italic> and <italic>X</italic>. For case 2, the relative max errors are 9.63% and 9.87%. From <xref ref-type="table" rid="T2">Table 2</xref>, average errors for <italic>R</italic> and <italic>X</italic> are 6.69% and 7.24%. The GAT-MMoE model can achieve the lowest error in most cases. The deviation comes from the Gaussian noise we add and as the resistance and reactance have different orders of magnitude, the errors are within the allowable range.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Identification results presented with true values and identification values. <bold>(A)</bold> line resistance of IEEE-14 system; <bold>(B)</bold> line reactance of IEEE-14 system; <bold>(C)</bold> line resistance of IEEE-33 system; <bold>(D)</bold> line reactance of IEEE-33 system.</p>
</caption>
<graphic xlink:href="fenrg-12-1485369-g007.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Identification errors in IEEE-14 system.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center">LR</th>
<th align="center">SVR</th>
<th align="center">FCN</th>
<th align="center">GAT-MMoE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Average error of <italic>R</italic>
</td>
<td align="center">0.3732</td>
<td align="center">0.0578</td>
<td align="center">0.0622</td>
<td align="center">0.0384</td>
</tr>
<tr>
<td align="center">Average error of <italic>X</italic>
</td>
<td align="center">0.3485</td>
<td align="center">0.0563</td>
<td align="center">0.0588</td>
<td align="center">0.0267</td>
</tr>
<tr>
<td align="center">Max error of <italic>R</italic>
</td>
<td align="center">0.1022</td>
<td align="center">0.0344</td>
<td align="center">0.0425</td>
<td align="center">0.0218</td>
</tr>
<tr>
<td align="center">Max error of <italic>X</italic>
</td>
<td align="center">0.1234</td>
<td align="center">0.0437</td>
<td align="center">0.0438</td>
<td align="center">0.0319</td>
</tr>
<tr>
<td align="center">Min error of <italic>R</italic>
</td>
<td align="center">0.4043</td>
<td align="center">0.0726</td>
<td align="center">0.0733</td>
<td align="center">0.0083</td>
</tr>
<tr>
<td align="center">Min error of <italic>X</italic>
</td>
<td align="center">0.3732</td>
<td align="center">0.0794</td>
<td align="center">0.0799</td>
<td align="center">0.0035</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Identification errors in IEEE-33 system.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center">LR</th>
<th align="center">SVR</th>
<th align="center">FCN</th>
<th align="center">GAT-MMoE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Average error of <italic>R</italic>
</td>
<td align="center">0.3763</td>
<td align="center">0.0609</td>
<td align="center">0.0652</td>
<td align="center">0.0569</td>
</tr>
<tr>
<td align="center">Average error of <italic>X</italic>
</td>
<td align="center">0.3815</td>
<td align="center">0.0593</td>
<td align="center">0.0618</td>
<td align="center">0.0426</td>
</tr>
<tr>
<td align="center">Max error of <italic>R</italic>
</td>
<td align="center">0.4173</td>
<td align="center">0.0756</td>
<td align="center">0.0763</td>
<td align="center">0.0638</td>
</tr>
<tr>
<td align="center">Max error of <italic>X</italic>
</td>
<td align="center">0.5489</td>
<td align="center">0.0824</td>
<td align="center">0.0929</td>
<td align="center">0.0612</td>
</tr>
<tr>
<td align="center">Min error of <italic>R</italic>
</td>
<td align="center">0.1552</td>
<td align="center">0.0434</td>
<td align="center">0.0495</td>
<td align="center">0.0168</td>
</tr>
<tr>
<td align="center">Min error of <italic>X</italic>
</td>
<td align="center">0.1264</td>
<td align="center">0.0467</td>
<td align="center">0.0568</td>
<td align="center">0.0153</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Compared with the baseline model, our proposed GAT-MMoE model demonstrates higher accuracy and better robustness considering measurement error. The results in <xref ref-type="table" rid="T1">Tables 1</xref>&#x2013;<xref ref-type="table" rid="T4">4</xref> indicate that the proposed GAT-MMoE method outperforms all other models in terms of RMSE and MAPE. Network-based methods generally show superior performance compared to linear regression and traditional machine learning methods, underscoring the value of graph attention learning in extracting high-quality features for DLP prediction. The superior performance of GAT-MMoE can be attributed to its effective utilization of related knowledge between neighboring nodes in the graph. Most baseline methods do not specifically address the issue of sparsity in their models, resulting in suboptimal performance. Our model leverages multiple data sources to construct the information network and employs a multi-task learning framework to address the specific task of predicting branch line parameters. Consequently, GAT-MMoE outperforms the selected baselines. Additionally, the model demonstrates good robustness when photovoltaic power supply is integrated into the system. Once the model training is completed, it can simultaneously predict all parameters of the power grid branch. Despite the lengthy training process, the method&#x2019;s robustness and accuracy compensate for this drawback. Moreover, the trained neural network model can be easily and rapidly deployed to the required locations, making it a practical solution for real-world applications.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Identification indexes compared with different baseline models in IEEE-14 system.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">R</th>
<th align="center">RMSE</th>
<th align="center">MAPE</th>
<th align="center">X</th>
<th align="center">RMSE</th>
<th align="center">MAPE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">LR</td>
<td align="center">0.8732</td>
<td align="center">0.7322</td>
<td align="center">LR</td>
<td align="center">0.4198</td>
<td align="center">0.3485</td>
</tr>
<tr>
<td align="center">SVR</td>
<td align="center">0.1924</td>
<td align="center">0.0845</td>
<td align="center">SVR</td>
<td align="center">0.2643</td>
<td align="center">0.0967</td>
</tr>
<tr>
<td align="center">FCN</td>
<td align="center">0.2494</td>
<td align="center">0.1412</td>
<td align="center">FCN</td>
<td align="center">0.0953</td>
<td align="center">0.0876</td>
</tr>
<tr>
<td align="center">GAT-MMoE</td>
<td align="center">0.0545</td>
<td align="center">0.0203</td>
<td align="center">GAT-MMoE</td>
<td align="center">0.0638</td>
<td align="center">0.0311</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Identification indexes compared with different baseline models in IEEE-33 system.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">R</th>
<th align="center">RMSE</th>
<th align="center">MAPE</th>
<th align="center">X</th>
<th align="center">RMSE</th>
<th align="center">MAPE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">LR</td>
<td align="center">0.9302</td>
<td align="center">0.1222</td>
<td align="center">LR</td>
<td align="center">0.5598</td>
<td align="center">0.5697</td>
</tr>
<tr>
<td align="center">SVR</td>
<td align="center">0.2521</td>
<td align="center">0.1245</td>
<td align="center">SVR</td>
<td align="center">0.2743</td>
<td align="center">0.1267</td>
</tr>
<tr>
<td align="center">FCN</td>
<td align="center">0.6578</td>
<td align="center">0.2612</td>
<td align="center">FCN</td>
<td align="center">0.3453</td>
<td align="center">0.2384</td>
</tr>
<tr>
<td align="center">GAT-MMoE</td>
<td align="center">0.0689</td>
<td align="center">0.0317</td>
<td align="center">GAT-MMoE</td>
<td align="center">0.0688</td>
<td align="center">0.0487</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From <xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref>, different we can find that our proposed GAT-MMoE model achieves the best identification results in most cases. The indexes of multi-task learning model are better than that of single task learning model. By comparing the machine learning methods, we can find that the LR method achieves very good identification results without noise, but when the input features contain disturbance and noise, the accuracy of branch parameter identification is greatly reduced. Deep learning method like FCN fail to balance accuracy between resistance and reactance, as the loss function is only simple addition. Also, the method is more inclined to the identification result of line resistance <italic>R</italic> and ignores the branch reactance <italic>X</italic>.</p>
<p>The influence of distributed photovoltaic access on the proposed identification method is further explored. We incorporated multiple distributed photovoltaic (PV) systems into the distribution network, with the power data of the PV sources derived from the Desert Knowledge Australia Solar Centre (<xref ref-type="bibr" rid="B7">DKA Solar Center, 2024</xref>) We integrated PV1 and PV2 at nodes 7 and 12 in the IEEE 14-bus system, and at nodes 22 and 33 in the IEEE 33-bus system. Based on the sampling frequency of every 15 min, the system node data is obtained by power flow calculation. After collecting system P, Q, and V data over multiple time profiles, we input them into the model for parameter identification. <xref ref-type="fig" rid="F8">Figure 8</xref> shows that the identification error increases when a distributed power supply is present in the network. This increase is due to the changes in power flow direction caused by the integration of distributed photovoltaics. However, the identification errors remain within the acceptable range, demonstrating the robustness of the method.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Identification results in IEEE-33 system with PV access: <bold>(A)</bold> line resistance MAPE of IEEE-33 system; <bold>(B)</bold> line reactance MAPE of IEEE-33 system.</p>
</caption>
<graphic xlink:href="fenrg-12-1485369-g008.tif"/>
</fig>
<p>For most existing distribution power system, the complexities and dynamic conditions present unique challenges and opportunities for parameter identification. Given that most existing power grid branch parameter identification methods are model-driven, resulting in low accuracy and poor reliability, our proposed model leverages a large volume of multi-source power grid operation data. It is constrained by the grid topology while integrating both local and global information. This approach allows for the comprehensive use of historical data to more accurately identify branch parameters, which can be then fed back to the power grid dispatch center. As a result, dispatch operators gain a clearer understanding of the changing trends in branch parameters, ensuring the safe and stable operation of the power grid.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>Parameter identification is crucial for distribution network scheduling and control, making it a significant research task. Current methods, which are primarily model-driven, are sensitive to data loss and noise. This paper introduces a novel line parameter identification method for medium-voltage distribution networks, considering the topology constraints of power network branches and being validated on IEEE14-M and IEEE33 systems. When there are too many layers of adjacent nodes, the global information tends to become similar, resulting in redundancy. By introducing an attention mechanism, the proposed method focuses on relatively important nodes and perform feature fusion on key branches and features. The proposed method consists of three components: graph generation, attention calculation, and multi-task prediction. The GAT module uses adaptive attention weights to flexibly model dependencies among different nodes. The MMoE algorithm addresses the coupling characteristics among multiple branches by utilizing multiple expert networks, thereby improving accuracy.</p>
<p>The method&#x2019;s effectiveness and robustness are validated through simulated grid tests, demonstrating improved results compared to traditional methods. Results show that the GAT-MMoE method achieves lower identification deviations 3.84% and 2.67% in IEEE14-M, and 5.69% and 4.26% in IEEE33 compared to the LR, SVR and FCN methods, achieving high prediction accuracy, good performance, and robustness against various types of noise.</p>
<p>Moreover, the GAT-MMoE method relies solely on nodal measurements of injected active power, reactive power, and voltage magnitude, streamlining the identification process without compromising accuracy. As smart grid technologies continue to evolve, data-driven deep learning approaches will play an increasingly important role in improving parameter identification in distribution networks. Future work will aim to extend this approach to a wider range of line parameters while addressing issues such as limited data availability and dynamic topology adaptation.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>XJ: Writing&#x2013;original draft, Writing&#x2013;review and editing. CZ: Formal Analysis, Methodology, Writing&#x2013;review and editing. QP: Validation, Visualization, Writing&#x2013;review and editing. LW: Investigation, Writing&#x2013;review and editing. BW: Writing&#x2013;original draft. YX: Software, Writing&#x2013;review and editing. KC: Resources, Writing&#x2013;review and editing. LF: Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. The authors received funding from the Technology Project of State Grid Co., Ltd. of Jiangsu Province (J2023018).</p>
</sec>
<ack>
<p>The authors would like to thank and acknowledge the Suzhou Power Supply Company, State Grid Jiangsu Electric Power Co., Ltd.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>Authors XJ, CZ, QP, LW, BW, YX, KC, and LF were employed by Suzhou Power Supply Company, State Grid Jiangsu Electric Power Co., Ltd.</p>
<p>The authors declare that this study received funding from the Technology Project of State Grid Co., Ltd. of Jiangsu Province .The funder had the following involvement in the study: study design, data collection and analysis, decision to publish, preparation of the manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abomazid</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>El-Taweel</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Farag</surname>
<given-names>H. E. Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Optimal energy management of hydrogen energy facility using integrated battery energy storage and solar photovoltaic systems</article-title>. <source>IEEE Trans. Sustain. Energy</source> <volume>13</volume>, <fpage>1457</fpage>&#x2013;<lpage>1468</lpage>. <pub-id pub-id-type="doi">10.1109/TSTE.2022.3161891</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Asprou</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kyriakides</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Identification and estimation of erroneous transmission line parameters using PMU measurements</article-title>,&#x201d; in <source>2018 IEEE power and energy Society general meeting (PESGM)</source>, <fpage>1</fpage>. <pub-id pub-id-type="doi">10.1109/PESGM.2018.8585896</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Equilibrium analysis of general N-population multi-strategy games for generation-side long-term bidding: An evolutionary game perspective</article-title>. <source>J. Clea. Product.</source> <volume>276</volume>, <fpage>124123</fpage>. <pub-id pub-id-type="doi">10.1016/j.jclepro.2020.124123</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>2PnS-EG: a general two-population n-strategy evolutionary game for strategic long-term bidding in a deregulated market under different market clearing mechanisms</article-title>. <source>Int. J. Electr. Power Energy Syst.</source> <volume>142</volume>, <fpage>108182</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2022.108182</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Behavioral decision-making in power demand-side response management: a multi-population evolutionary game dynamics perspective</article-title>. <source>Int. J. Electr. Power Energy Syst.</source> <volume>276</volume>, <fpage>106743</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2020.106743</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A new generation of AI: a review and perspective on machine learning technologies applied to smart energy and electric power systems</article-title>. <source>Int. J. Energy Res.</source> <volume>43</volume>, <fpage>1928</fpage>&#x2013;<lpage>1973</lpage>. <pub-id pub-id-type="doi">10.1002/er.4333</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Cipolla</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gal</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kendall</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Multi-task learning using uncertainty to weigh losses for scene geometry and semantics</article-title>,&#x201d; in <conf-name>2018 IEEE/CVF Conference on computer Vision and pattern Recognition</conf-name>, (<conf-loc>Salt lake City, UT, USA</conf-loc>: <publisher-name>IEEE</publisher-name>), <conf-date>18-23 June 2018</conf-date>, <fpage>7482</fpage>&#x2013;<lpage>7491</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2018.00781</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<collab>DKA Solar Center</collab> (<year>2024</year>). <source>General: Desert Knowledge Australia Centre</source>. <publisher-loc>Alice Springs, Australia</publisher-loc>: <publisher-name>Download Data. Alice Springs</publisher-name>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="http://dkasolarcentre.com.au/download">http://dkasolarcentre.com.au/download</ext-link> (Accessed June 20, 2024)</comment>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eigen</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ranzato</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Learning factored representations in a deep mixture of experts</article-title>. <source>arXiv.Org</source>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1312.4314v3">https://arxiv.org/abs/1312.4314v3</ext-link> (Accessed July 10, 2024)</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.1312.4314</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fern&#xe1;ndez-Delgado</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sirsat</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Cernadas</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Alawadi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Barro</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Febrero-Bande</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An extensive experimental survey of regression methods</article-title>. <source>Neural Netw.</source> <volume>111</volume>, <fpage>11</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2018.12.010</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jacobs</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Jordan</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>Nowlan</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>1991</year>). <article-title>Adaptive mixtures of local experts</article-title>. <source>Neural comput.</source> <volume>3</volume>, <fpage>79</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1162/neco.1991.3.1.79</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Terzija</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Measurement-based transmission line parameter estimation with adaptive data selection Scheme</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>9</volume>, <fpage>5764</fpage>&#x2013;<lpage>5773</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2017.2696619</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Vittal</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Blasch</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Distribution grid topology and parameter estimation using deep-shallow neural network with physical consistency</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>15</volume>, <fpage>655</fpage>&#x2013;<lpage>666</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2023.3278702</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Reverse identification method of line parameters in distribution network with multi-T nodes based on partial measurement data</article-title>. <source>Electr. Pow. Syst. Res.</source> <volume>204</volume>, <fpage>107691</fpage>. <pub-id pub-id-type="doi">10.1016/j.epsr.2021.107691</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Event-triggered-based distributed cooperative energy management for multienergy systems</article-title>. <source>IEEE Trans. Ind. Inf.</source> <volume>15</volume>, <fpage>2008</fpage>&#x2013;<lpage>2022</lpage>. <pub-id pub-id-type="doi">10.1109/TII.2018.2862436</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Abur</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Strategic use of synchronized phasor measurements to improve network parameter error detection</article-title>. <source>IEEE Trans. Smart Grid</source> <volume>9</volume>, <fpage>5281</fpage>&#x2013;<lpage>5290</lpage>. <pub-id pub-id-type="doi">10.1109/TSG.2017.2686095</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chi</surname>
<given-names>E. H.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Modeling task relationships in multi-task learning with multi-gate mixture-of-experts</article-title>,&#x201d; in <conf-name>Proceedings of the 24th ACM SIGKDD International Conference on knowledge Discovery &#x26; data mining</conf-name>, <fpage>1930</fpage>&#x2013;<lpage>1939</lpage>. <pub-id pub-id-type="doi">10.1145/3219819.3220007</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pegoraro</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Brady</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Castello</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Muscas</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>von Meier</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019a</year>). <article-title>Compensation of systematic measurement errors in a PMU-based monitoring system for electric distribution grids</article-title>. <source>IEEE Trans. Instrum. Meas.</source> <volume>68</volume>, <fpage>3871</fpage>&#x2013;<lpage>3882</lpage>. <pub-id pub-id-type="doi">10.1109/TIM.2019.2908703</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pegoraro</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Brady</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Castello</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Muscas</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>von Meier</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019b</year>). <article-title>Line impedance estimation based on synchrophasor measurements for power distribution systems</article-title>. <source>IEEE Trans. Instrum. Meas.</source> <volume>68</volume>, <fpage>1002</fpage>&#x2013;<lpage>1013</lpage>. <pub-id pub-id-type="doi">10.1109/TIM.2018.2861058</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pegoraro</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Castello</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Muscas</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Brady</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>von Meier</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Handling instrument transformers and PMU errors for the estimation of line parameters in distribution grids</article-title>,&#x201d; in <source>2017 IEEE international workshop on applied measurements for power systems (AMPS)</source>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1109/AMPS.2017.8078339</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pegoraro</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Sitzia</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Solinas</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Sulis</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>PMU-based estimation of systematic measurement errors, line parameters, and tap changer ratios in three-phase power systems</article-title>. <source>IEEE Trans. Instrum. Meas.</source> <volume>71</volume>, <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1109/TIM.2022.3165247</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Puddu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Brady</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Muscas</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Pegoraro</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Von Meier</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>PMU-based technique for the estimation of line parameters in three-phase electric distribution grids</article-title>,&#x201d; in <source>2018 IEEE 9th international workshop on applied measurements for power systems (AMPS)</source>, <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/AMPS.2018.8494886</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Mirhoseini</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Maziarz</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Davis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Outrageously large neural networks: the sparsely-gated mixture-of-experts layer</article-title>. <pub-id pub-id-type="doi">10.48550/arXiv.1701.06538</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ling</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chu</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Early anomaly detection and localisation in distribution network: a data-driven approach</article-title>. <source>IET Gener. Transm. Distrib.</source> <volume>14</volume>, <fpage>3814</fpage>&#x2013;<lpage>3825</lpage>. <pub-id pub-id-type="doi">10.1049/iet-gtd.2019.1790</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>Z. .</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Line parameter, topology and phase estimation in three-phase distribution networks with non-&#x3bc;PMUs (2024)</article-title>. <source>Int. J. Electr. Power Energy Syst.</source> <volume>155</volume>, <fpage>109658</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijepes.2023.109658</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Cobben</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gibescu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>van den Brom</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Colangelo</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Rietveld</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Medium voltage line parameter estimation using synchrophasor data: a step towards dynamic line rating</article-title>,&#x201d; in <source>2018 IEEE power and energy society general meeting (PESGM)</source>, <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/PESGM.2018.8586111</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A classification identification method based on phasor measurement for distribution line parameter identification under insufficient measurements conditions</article-title>. <source>IEEE Access</source> <volume>7</volume>, <fpage>158732</fpage>&#x2013;<lpage>158743</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2019.2950461</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thurner</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Scheidler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sch&#xe4;fer</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Menke</surname>
<given-names>J.-H.</given-names>
</name>
<name>
<surname>Dollichon</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Meier</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Pandapower&#x2014;an open-source python tool for convenient modeling, analysis, and optimization of electric power systems</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>33</volume>, <fpage>6510</fpage>&#x2013;<lpage>6521</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2018.2829021</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Velickovic</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cucurull</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Casanova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Romero</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lio&#x27;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Graph attention networks</article-title>. <source>ArXiv</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1710.10903</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Estimate three-phase distribution line parameters with physics-informed graphical learning method</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>37</volume>, <fpage>3577</fpage>&#x2013;<lpage>3591</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2021.3134952</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Method to measure the unbalance of the multiple-circuit transmission lines on the same tower and its applications</article-title>. <source>IET Gener. Transm. Distrib.</source> <volume>10</volume>, <fpage>2050</fpage>&#x2013;<lpage>2057</lpage>. <pub-id pub-id-type="doi">10.1049/iet-gtd.2015.0979</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Parameter identification in power transmission systems based on graph convolution network</article-title>. <source>IEEE Trans. Power Del.</source> <volume>37</volume>, <fpage>3155</fpage>&#x2013;<lpage>3163</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRD.2021.3124528</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>MFAGCN: a new framework for identifying power grid branch parameters</article-title>. <source>Electr. Pow. Syst. Res.</source> <volume>207</volume>, <fpage>107855</fpage>. <pub-id pub-id-type="doi">10.1016/j.epsr.2022.107855</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Distribution line parameter estimation driven by probabilistic data fusion of D-PMU and AMI</article-title>. <source>IET Gener. Transm. Distrib.</source> <volume>20</volume>, <fpage>2883</fpage>&#x2013;<lpage>2892</lpage>. <pub-id pub-id-type="doi">10.1049/gtd2.12224</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Hybrid policy-based reinforcement learning of adaptive energy management for the energy transmission-constrained island group</article-title>. <source>IEEE Trans. Ind. Inf.</source> <volume>19</volume>, <fpage>10751</fpage>&#x2013;<lpage>10762</lpage>. <pub-id pub-id-type="doi">10.1109/TII.2023.3241682</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>N.-C.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>M.-F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Distribution feeder parameter estimation without synchronized phasor measurement by using radial basis function neural networks and multi-run optimization method</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>2869</fpage>&#x2013;<lpage>2879</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2021.3140123</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Rajagopal</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>PaToPa: a data-driven parameter and topology joint estimation framework in distribution grids</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>33</volume>, <fpage>4335</fpage>&#x2013;<lpage>4347</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2017.2778194</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Rajagopal</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>PaToPaEM: a data-driven parameter and topology joint estimation framework for time-varying system in distribution grids</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>34</volume>, <fpage>1682</fpage>&#x2013;<lpage>1692</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2018.2888619</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Interval state estimation with uncertainty of distributed generation and line parameters in unbalanced distribution systems</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>35</volume>, <fpage>762</fpage>&#x2013;<lpage>772</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2019.2926445</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Abur</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Improvements in network parameter error identification via synchronized phasors</article-title>. <source>IEEE Trans. Power Syst.</source> <volume>25</volume>, <fpage>44</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1109/TPWRS.2009.2030274</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zou</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Intelligent identification of power grid parameters based on dynamic weighting</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>135</volume>, <fpage>108822</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2024.108822</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>