<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Phys.</journal-id>
<journal-title>Frontiers in Physics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Phys.</abbrev-journal-title>
<issn pub-type="epub">2296-424X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1253953</article-id>
<article-id pub-id-type="doi">10.3389/fphy.2023.1253953</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Topological data analysis of Chinese stocks&#x2019; dynamic correlations under major public events</article-title>
<alt-title alt-title-type="left-running-head">Guo et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphy.2023.1253953">10.3389/fphy.2023.1253953</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Guo</surname>
<given-names>Hongfeng</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2370411/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ming</surname>
<given-names>Ziwei</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2346930/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xing</surname>
<given-names>Bing</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff>
<institution>School of Statistics and Mathematics</institution>, <institution>Shandong University of Finance and Economics</institution>, <addr-line>Jinan</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2283892/overview">Marcin W&#x105;torek</ext-link>, Cracow University of Technology, Poland</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1305588/overview">Max Menzies</ext-link>, Tsinghua University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2371210/overview">Nick James</ext-link>, The University of Melbourne, Australia</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Hongfeng Guo, <email>guohongfeng@sdufe.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>08</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>11</volume>
<elocation-id>1253953</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>08</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Guo, Ming and Xing.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Guo, Ming and Xing</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Topological data analysis has been acknowledged as one of the most successful mathematical data analytic methodologies in many fields. Additionally, it has also been gradually applied in financial time series analysis and proved effective in exploring the topological features of such data. We select 100 stocks from China&#x2019;s markets and construct point cloud data for topological data analysis. We detect critical dates from the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms of the persistence landscapes. Our results reveal the dates are highly consistent with the transition time of some major events in the sample period. We compare the correlations and statistical properties of stocks before and during the events via complex networks to describe the markets&#x2019; situation. The strength and variation of links among stocks are clearly different during the major events. We also investigate the neighborhood features of stocks from topological perspectives. This helps identify the important stocks and explore their situations under each event. Finally, we cluster the stocks based on the neighborhood features, which exhibit the heterogeneity impact on stocks of the different events. Our work demonstrates that topological data analysis has strong applicability in the dynamic correlations of stocks.</p>
</abstract>
<kwd-group>
<kwd>topological data analysis</kwd>
<kwd>persistence landscape</kwd>
<kwd>L<sup>p</sup>-norm</kwd>
<kwd>complex network</kwd>
<kwd>major public event</kwd>
<kwd>systemic financial risk</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Social Physics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>In recent years, major public events have frequently occurred, and they can easily bring substantial and negative impacts to the financial markets, leading to dramatic fluctuations in stock prices. For example, the coronavirus disease 2019 (COVID-19) has swept across the world, rapidly spreading to more than 200 countries and becoming a global public health and economic catastrophe [<xref ref-type="bibr" rid="B1">1</xref>]. Under such situations, people are more cautious about investing, resulting in a significant decrease in market liquidity [<xref ref-type="bibr" rid="B2">2</xref>]. Tracking the dynamic market changes and identifying critical information is vital to prevent systemic financial risks. Various methods and models have been applied to help identify the risks or identify early warning signals.</p>
<p>Before 2008, the early warning indicator method was one of the most common for measuring systemic financial risks, being simple, clear, practical, and effective. However, it only investigates static conditions of financial systems. The risk contagion effect, negative external characteristics, and system correlations remain unknown under this method [<xref ref-type="bibr" rid="B3">3</xref>]. The composite index method significantly compensates for this deficiency, helping to dynamically reflect the comprehensive situation of the risks of the financial system. However, the various indicator levels are all artificially weighted, which separates the dynamic evolution process and inevitably generates prediction errors. After the advent of the 2008 global financial crisis, scholars began to pay more attention to the correlations and contagions of risks within the financial systems, and financial risk spillover indices were widely used [<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B5">5</xref>]. The related models and methods were gradually enriched [<xref ref-type="bibr" rid="B6">6</xref>&#x2013;<xref ref-type="bibr" rid="B10">10</xref>]. Recently, multiple models have been commonly used to comprehensively consider financial risk [<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>]. However, most of these studies often use a representative stock index for each country. For example, the CSI 300 Index is most commonly used in China. Most studies directly use the stock indices to measure risks [<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B14">14</xref>]. However, the indices cannot describe the changes in the stocks&#x2019; internal structures more comprehensively and accurately. The components of the CSI 300 Index are adjusted every 6&#xa0;months in principle, and the weights are calculated based on the free float. Although the weights change dynamically, it is not easy to extract the internal structure of relatively similar stocks. For example, if two components of a stock index have exactly the same weighting, and if the rise of one is exactly as great as that of the other, then the change in the index is hardly captured. Therefore, we focus on China&#x2019;s markets and attempt to identify a new indicator that can accurately portray the dynamic evolution of financial risks and effectively capture the risk changes in the financial system under major events. This indicator can help obtain early warning signals of systemic risks in China&#x2019;s stock market.</p>
<p>We apply topological data analysis (TDA) by constructing point clouds data of stock returns over the past 20 years and calculating <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms of the persistence landscapes of the data to obtain a new indicator with a topological perspective. TDA provides an increasing number of methods to explore topological features of multifarious data [<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B16">16</xref>], especially for big data. Carlsson [<xref ref-type="bibr" rid="B16">16</xref>] systematically introduces the theoretical basis of simple homology, persistent homology, and complex shape; among which, the advantage of persistent homology is that it can accurately describe the whole picture of the data without dimension reduction and exhibit key topological features such as the number of clusters and ring structures in the data. Bubenik [<xref ref-type="bibr" rid="B17">17</xref>] adds the concept of persistent landscapes to provide reasonable quantification of persistent homology. Furthermore, the construction of the persistence landscapes toolbox [<xref ref-type="bibr" rid="B18">18</xref>] provided important support for the application. In their investigation of financial markets based on persistent landscapes, Gidea [<xref ref-type="bibr" rid="B19">19</xref>] documents that <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms display a significant trend during market crashes. Guo et al. [<xref ref-type="bibr" rid="B20">20</xref>] demonstrate that TDA works well in analyzing financial crises. With the continuous deepening of research, topology is gradually becoming widely used in financial markets [<xref ref-type="bibr" rid="B21">21</xref>, <xref ref-type="bibr" rid="B22">22</xref>]. Therefore, we conduct TDA on stocks chosen from China&#x2019;s market to investigate topological features, especially situations before and during a major public event. We identify the critical dates when <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms decline significantly, which indicates an obvious increase in the stocks&#x2019; overall correlations. We then construct complex networks for individual stocks to provide reasonable explanations of correlations within the stock market. We derive the neighborhood norms for each stock and use K-means to cluster the stocks. We also compare the changes in the classification of individual stocks before and after the critical dates.</p>
<p>The procedure implemented in this study is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>. We use TDA to analyze the topological features of the stocks from a new perspective and provide new indicators to capture the critical changes in the data. We creatively select 30 stocks with the highest correlations of each stock as its neighborhood(s) and calculate the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms of every such clique to investigate the neighborhood feature(s) of each stock, which can reflect the position and importance of each stock among the whole. We consider the norm as a higher-order attribute for each stock and cluster them accordingly. We then comprehensively analyze the dynamic evolution of the correlations among stocks in various industries during major public events.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Flowchart of the procedure implemented. There are two parts. The first part involves data collection, processing, and point cloud construction. The second concerns TDA-related computations and empirical study.</p>
</caption>
<graphic xlink:href="fphy-11-1253953-g001.tif"/>
</fig>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Data collection and processing</title>
<p>We focus our research on China&#x2019;s stock markets. First, according to the latest Standard Industrial Classification (SIC system), we randomly selected 150 A-share stocks in various industries and collected the closing prices of Shanghai and Shenzhen A-share component stocks between 2005/01/01 and 2022/03/31 from the Wind Financial Database. Considering factors such as long-term suspension and delisting, we cleaned the data and derived a sample of 100 stocks to examine their performance in terms of various stock market indicators during typical major events in the sample period.</p>
<p>There is a small amount of missing data among the daily closing price series of stocks. We interpolate the closing price and finally obtain that of 4,189 trading days. We use their daily log returns; that is, <italic>r</italic>
<sub>
<italic>i</italic>
</sub>(<italic>t</italic>) &#x3d; <italic>lnP</italic>
<sub>
<italic>i</italic>
</sub>(<italic>t</italic>) &#x2212; <italic>lnP</italic>
<sub>
<italic>i</italic>
</sub>(<italic>t</italic> &#x2212; 1), <italic>i</italic> &#x2208; 1, 2, &#x2026;, 100, where <italic>P</italic>
<sub>
<italic>i</italic>
</sub>(<italic>t</italic>) is the closing price of the index on trading day&#xa0;<italic>t</italic>.</p>
<p>We applied the sliding window method to process the data as follows. For each trading day&#xa0;<italic>t</italic>, we selected the daily return series (<italic>r</italic>
<sub>
<italic>i</italic>
</sub>(<italic>t</italic> &#x2212; <italic>&#x3c9;</italic> &#x2b; 1), <italic>r</italic>
<sub>
<italic>i</italic>
</sub>(<italic>t</italic> &#x2212; <italic>&#x3c9;</italic> &#x2b; 2), &#x2026;, <italic>r</italic>
<sub>
<italic>i</italic>
</sub>(<italic>t</italic>)) and (<italic>r</italic>
<sub>
<italic>j</italic>
</sub>(<italic>t</italic> &#x2212; <italic>&#x3c9;</italic> &#x2b; 1), <italic>r</italic>
<sub>
<italic>j</italic>
</sub>(<italic>t</italic> &#x2212; <italic>&#x3c9;</italic> &#x2b; 2), &#x2026;, <italic>r</italic>
<sub>
<italic>j</italic>
</sub>(<italic>t</italic>)) for each pair of stocks (<italic>i</italic>, <italic>j</italic>) based on the sliding window size <italic>&#x3c9;</italic> (i.e., from trading day&#xa0;<italic>t</italic> &#x2212; <italic>&#x3c9;</italic> &#x2b; 1 to trading day&#xa0;<italic>t</italic>). We then obtained the corresponding correlation coefficients <italic>C</italic>
<sub>
<italic>t</italic>
</sub>(<italic>i</italic>, <italic>j</italic>) and transformed into the distance measure <italic>d</italic>
<sub>
<italic>t</italic>
</sub>(<italic>i</italic>, <italic>j</italic>) using <inline-formula id="inf1">
<mml:math id="m1">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s2-2">
<title>2.2 Background of TDA</title>
<sec id="s2-2-1">
<title>2.2.1 Persistent homology</title>
<p>To better understand persistent homology, imagine that we have a point cloud. Furthermore, suppose there is a parameter that defines the radius of a ball centered on each data point, and as the parameter increases, each ball gradually comes into contact with the other balls. These processes of ball contact enable the emergence of unique topological features, thus providing a new perspective on financial time series and other data. TDA is constructed as a filter of complex shapes while calculating persistent homology with the point cloud data to sort a certain resolution parameter. It is characterized by the fact that the data is not only retained in the original high-dimensional space, but also the number of clusters and ring structures present is identified, and the whole process does not require visualization of the data.</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Simplicial complexes</title>
<p>A simplex is defined on a finite vertex set. For example, a <italic>k</italic>-simplex is a polyhedron with <italic>k</italic> &#x2b; 1 vertices and <italic>k</italic> &#x2b; 1 faces. A simplicial complex <italic>K</italic> is a set of simplexes and satisfies the following two conditions: (1) Any face of any simplex in <italic>K</italic> still belongs to <italic>K</italic>. (2) The intersection of any two simplexes <italic>&#x3ba;</italic>
<sub>1</sub>, <italic>&#x3ba;</italic>
<sub>2</sub> in <italic>K</italic> is either the empty set or one of the common surfaces of both. The most common simplicial complexes are Vietoris&#x2014;Rips, &#x10c;ech, witness, and Alpha complexes, among others. We use the Vietoris-Rips complexes because these approximate the more exact &#x10c;ech complexes but are more efficient to calculate [<xref ref-type="bibr" rid="B22">22</xref>].</p>
<p>Suppose we have a point cloud <italic>A</italic> &#x3d; {<italic>x</italic>
<sub>1</sub>, <italic>x</italic>
<sub>2</sub>, &#x2026;, <italic>x</italic>
<sub>
<italic>n</italic>
</sub>} in a topological space <italic>X</italic> endowed with a metric <italic>d</italic>. Denote the <italic>&#x25b;</italic>-ball of each point <italic>x</italic> &#x2208; <italic>X</italic> by <italic>B</italic>
<sub>
<italic>&#x25b;</italic>
</sub> &#x3d; {<italic>y</italic> &#x2208; <italic>X</italic>: <italic>d</italic>(<italic>y</italic>, <italic>x</italic>) &#x3c; <italic>&#x25b;</italic>, <italic>y</italic> &#x2260; <italic>x</italic>} for any <italic>&#x25b;</italic> &#x3e; 0. A <italic>Vietoris&#x2013;Rips complex</italic> <italic>VR</italic>(<italic>A</italic>, <italic>&#x25b;</italic>) w.r.t the positive value <italic>&#x25b;</italic> is the simplicial complex whose vertices set is <italic>A</italic> and where <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> spans a <italic>k</italic>-simplex if and only if <inline-formula id="inf3">
<mml:math id="m3">
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:math>
</inline-formula> for any 0 &#x2264; <italic>j</italic>, <italic>l</italic> &#x2264; <italic>k</italic>. Note that in <italic>VR</italic>(<italic>A</italic>, <italic>&#x25b;</italic>), two points <italic>x</italic>
<sub>
<italic>i</italic>
</sub> and <italic>x</italic>
<sub>
<italic>j</italic>
</sub> are connected when <italic>d</italic>(<italic>x</italic>
<sub>
<italic>i</italic>
</sub>, <italic>x</italic>
<sub>
<italic>j</italic>
</sub>) &#x3c; <italic>&#x25b;</italic>. Moreover, it is not difficult to see that a <italic>k</italic>-simplex may merge into a new higher dimensional simplex with the increasing of the value <italic>&#x25b;</italic>. For the concepts of <italic>k</italic>-simplex and simplicial complex, please refer to Munkres [<xref ref-type="bibr" rid="B23">23</xref>].</p>
</sec>
<sec id="s2-2-3">
<title>2.2.3 Barcodes and persistence diagrams</title>
<p>The persistence barcode is the most direct way to characterize a complex and quantify its change with the increasing value <italic>&#x25b;</italic>. A <italic>barcode</italic> [<xref ref-type="bibr" rid="B24">24</xref>] is a graphical representation of a collection of horizontal line segments {<italic>I</italic>
<sub>
<italic>j</italic>
</sub>: <italic>j</italic> &#x2208; <italic>J</italic>} in a plane with each interval as the life of a topological hole, that is, a homology class. For example, an interval <italic>I</italic>
<sub>
<italic>j</italic>
</sub> &#x3d; [<italic>a</italic>
<sub>
<italic>j</italic>
</sub>, <italic>b</italic>
<sub>
<italic>j</italic>
</sub>] in a 0-dim barcode corresponds to a connected component in the complex with the meaning that the component emerges when <italic>&#x25b;</italic> &#x3d; <italic>a</italic>
<sub>
<italic>j</italic>
</sub> and vanishes when <italic>&#x25b;</italic> &#x3d; <italic>b</italic>
<sub>
<italic>j</italic>
</sub>; an interval in a 1-dim barcode corresponds to an independent loop in the complex with the endpoints as the value when the loop emerges and vanishes. However, it is not sufficient to analyze the shape of the data only by the barcodes; we need other ways to convert the barcodes to computable properties. A natural representation of a barcode called a <italic>persistence diagram</italic> (see Cohen-Steiner et al. [<xref ref-type="bibr" rid="B25">25</xref>]) is the set {(<italic>a</italic>
<sub>
<italic>j</italic>
</sub>, <italic>b</italic>
<sub>
<italic>j</italic>
</sub>): <italic>j</italic> &#x2208; <italic>J</italic>}. Note that each (<italic>a</italic>
<sub>
<italic>j</italic>
</sub>, <italic>b</italic>
<sub>
<italic>j</italic>
</sub>) here is a 2-dim point, where <italic>a</italic>
<sub>
<italic>j</italic>
</sub> and <italic>b</italic>
<sub>
<italic>j</italic>
</sub> are the endpoints of the interval <italic>I</italic>
<sub>
<italic>j</italic>
</sub> in the barcode. We can compare and calculate the difference between the diagrams by bottleneck distance and degree <italic>p</italic> Wasserstein distance (see Cohen-Steiner et al. [<xref ref-type="bibr" rid="B25">25</xref>], Gidea and Katz [<xref ref-type="bibr" rid="B19">19</xref>], respectively). They induce a metric on the space of persistence diagrams.</p>
</sec>
<sec id="s2-2-4">
<title>2.2.4 Persistence landscapes and <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms</title>
<p>To grasp the continual change of the data shapes, persistence landscapes, and their <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms introduced in Bubenik [<xref ref-type="bibr" rid="B17">17</xref>] improve the method effectively. Applying them to the complexes, we can observe the shapes&#x2019; continual change of the point cloud data. Furthermore, persistence landscapes maintain robustness as diagrams under perturbations of the data. Briefly, the <italic>persistent landscapes</italic> are the piecewise functions corresponding to any point (<italic>a</italic>
<sub>
<italic>j</italic>
</sub>, <italic>b</italic>
<sub>
<italic>j</italic>
</sub>) in the persistent graph as follows,<disp-formula id="e1">
<mml:math id="m4">
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mo>&#x2212;</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>Then, we define<disp-formula id="e2">
<mml:math id="m5">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>:</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(2)</label>
</disp-formula>where <italic>kmax</italic> denotes the <italic>k</italic>-th largest element for <inline-formula id="inf4">
<mml:math id="m6">
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. According to Bubenik [<xref ref-type="bibr" rid="B17">17</xref>], a <italic>persistence landscape</italic> is a sequence of functions <inline-formula id="inf5">
<mml:math id="m7">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="double-struck">R</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:math>
</inline-formula>, and <italic>&#x3bb;</italic>
<sub>
<italic>k</italic>
</sub> is called the <italic>k</italic>-th persistence landscape function. The <italic>critical points</italic> of <italic>&#x3bb;</italic>
<sub>
<italic>k</italic>
</sub> are those values of <italic>x</italic> at which the slope changes. The set of critical points of the persistence landscape <italic>&#x3bb;</italic> is the union of the sets of critical points of the functions <italic>&#x3bb;</italic>
<sub>
<italic>k</italic>
</sub>. Suppose a persistence diagram <italic>D</italic> &#x3d; {(1, 9), (2, 8), (4, 12), (5, 7)}, we then have <italic>&#x3bb;</italic>
<sub>
<italic>k</italic>
</sub>, (<italic>k</italic> &#x3d; 1, 2, 3, 4) and present the corresponding broken lines in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Persistence landscapes of an example, i.e., images of the functions <italic>&#x3bb;</italic>
<sub>1</sub>, <italic>&#x3bb;</italic>
<sub>2</sub>, <italic>&#x3bb;</italic>
<sub>3</sub>, and <italic>&#x3bb;</italic>
<sub>4</sub> from top to bottom. We calculate them using Eqs <xref ref-type="disp-formula" rid="e1">1</xref>, <xref ref-type="disp-formula" rid="e2">2</xref>. The critical points of the persistence landscape <inline-formula id="inf6">
<mml:math id="m8">
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2,3,4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> are (1, 0), (9, 0), (5, 4), (2, 0), (8, 0), (5, 3), (4, 0), (12, 0), (8, 4), (6.5, 2.5), (6, 2), (5, 0), (7, 0) and (6, 1).</p>
</caption>
<graphic xlink:href="fphy-11-1253953-g002.tif"/>
</fig>
<p>Persistence landscapes form a subset of the Banach space <inline-formula id="inf7">
<mml:math id="m9">
<mml:msup>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="double-struck">N</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. For two persistence landscapes <inline-formula id="inf8">
<mml:math id="m10">
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula id="inf9">
<mml:math id="m11">
<mml:mi>&#x3b7;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, it is clear that (<italic>&#x3bb;</italic> &#x2b; <italic>&#x3b7;</italic>)<sub>
<italic>k</italic>
</sub>(<italic>x</italic>) &#x3d; <italic>&#x3bb;</italic>
<sub>
<italic>k</italic>
</sub>(<italic>x</italic>) &#x2b; <italic>&#x3b7;</italic>
<sub>
<italic>k</italic>
</sub>(<italic>x</italic>) and (<italic>c</italic> &#x22c5; <italic>&#x3bb;</italic>)<sub>
<italic>k</italic>
</sub>(<italic>x</italic>) &#x3d; <italic>c</italic> &#x22c5; <italic>&#x3bb;</italic>
<sub>
<italic>k</italic>
</sub>(<italic>x</italic>) for all <inline-formula id="inf10">
<mml:math id="m12">
<mml:mi>x</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:math>
</inline-formula>, and <inline-formula id="inf11">
<mml:math id="m13">
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="double-struck">N</mml:mi>
</mml:math>
</inline-formula>. It becomes a Banach space when the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms of a persistence landscape <inline-formula id="inf12">
<mml:math id="m14">
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> are defined in the following way for 1 &#x2264; <italic>p</italic> &#x2264; <italic>&#x221e;</italic>,<disp-formula id="e3">
<mml:math id="m15">
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf13">
<mml:math id="m16">
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> denotes the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms of <italic>&#x3bb;</italic>
<sub>
<italic>k</italic>
</sub>, that is, <inline-formula id="inf14">
<mml:math id="m17">
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mo>&#x222b;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> with respect to the Lebesgue measure. In summary, the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms are based on the simplicial complex of high-dimensional point clouds, and the values are calculated based on distances and the structures of the point clouds. The smaller the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms correspond, the closer the distances. Note that a barcode corresponds to an interval <italic>I</italic>
<sub>
<italic>j</italic>
</sub> &#x3d; [<italic>a</italic>
<sub>
<italic>j</italic>
</sub>, <italic>b</italic>
<sub>
<italic>j</italic>
</sub>], and an interval corresponds to a broken line (i.e., the <inline-formula id="inf15">
<mml:math id="m18">
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> function) in persistence landscapes. In other words, there are as many barcodes as there are <inline-formula id="inf16">
<mml:math id="m19">
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> functions. Additionally, the number of <italic>&#x3bb;</italic>
<sub>
<italic>k</italic>
</sub> is always less than or equal to the number of <inline-formula id="inf17">
<mml:math id="m20">
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> function.</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Background and topological measures of complex networks</title>
<p>To establish a threshold network, we first construct an adjacency matrix <italic>A</italic> &#x3d; (<italic>A</italic>(<italic>u</italic>, <italic>v</italic>)) based on the distance matrix <italic>D</italic> &#x3d; (<italic>d</italic>(<italic>u</italic>, <italic>v</italic>)). When <italic>d</italic>(<italic>u</italic>, <italic>v</italic>) is less than or equal to a given threshold <italic>&#x3b8;</italic>, the corresponding element <italic>A</italic>(<italic>u</italic>, <italic>v</italic>) &#x3d; 1 in the adjacency matrix, and then the vertices <italic>u</italic> and <italic>v</italic> are connected by an edge (also called correlated); otherwise, <italic>A</italic>(<italic>u</italic>, <italic>v</italic>) &#x3d; 0, and thus there is no edge connecting <italic>u</italic> and <italic>v</italic> directly. Considering the importance of the adjacency matrix, we present its definition below,<disp-formula id="e4">
<mml:math id="m21">
<mml:mi>A</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3e;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>Note that the edges in the network have neither direction nor weight, and we refer to it as a non-directional and unweighted network. Such a complex network constructed in the above way is called a <italic>threshold network</italic>. To observe the features of such networks, we introduce four topological properties, including degree, density, clustering coefficient, and average path length.</p>
<p>Suppose the network contains <italic>N</italic> vertices. <italic>The degree of a vertex</italic> <italic>u</italic> is the number of edges connected to that vertex. The average degree of a network is the average of degrees for all vertices in the network [<xref ref-type="bibr" rid="B26">26</xref>], referred to as the <italic>degree of the network</italic>. Denote by <italic>M</italic> the actual number of edges. The maximum possible number of edges is obviously <inline-formula id="inf18">
<mml:math id="m22">
<mml:mfrac>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:math>
</inline-formula> in the network. Hence, the <italic>network density</italic> is defined and denoted by<disp-formula id="e5">
<mml:math id="m23">
<mml:mi>&#x3c1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>Denote by <italic>k</italic>
<sub>
<italic>u</italic>
</sub> the degree of the vertex <italic>u</italic>, that is, there are <italic>k</italic>
<sub>
<italic>u</italic>
</sub> vertices connecting the vertex <italic>u</italic>. <italic>E</italic>
<sub>
<italic>u</italic>
</sub> is the number of edges between these <italic>k</italic>
<sub>
<italic>u</italic>
</sub> vertices. The <italic>clustering coefficient of vertex</italic> <italic>u</italic>, denoted by <italic>C</italic>
<sub>
<italic>u</italic>
</sub>, is then defined in the following way when <italic>k</italic>
<sub>
<italic>u</italic>
</sub> &#x2265; 2,<disp-formula id="e6">
<mml:math id="m24">
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>When <italic>k</italic>
<sub>
<italic>u</italic>
</sub> is equal to 0 or 1, the clustering coefficient <italic>C</italic>
<sub>
<italic>u</italic>
</sub> of vertex <italic>u</italic> is specified as 0. The <italic>clustering coefficient of the network</italic> is the average clustering coefficient of all vertices in the network, that is, <inline-formula id="inf19">
<mml:math id="m25">
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mi mathvariant="normal">&#x3a3;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>.</p>
<p>In the network, the number of edges contained in the shortest path from the vertex <italic>u</italic> to the vertex <italic>v</italic> is referred to as the <italic>shortest path length between</italic> <italic>u</italic> <italic>and</italic> <italic>v</italic>, denoted by <italic>D</italic>
<sub>
<italic>uv</italic>
</sub>. The <italic>average path length of the network</italic>, denoted by <italic>L</italic>, is defined as the average of the shortest path length between each pair of vertices in the network, that is,<disp-formula id="e7">
<mml:math id="m26">
<mml:mi>L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>By the definitions of the above four topological properties, we know that the larger they are, the closer the vertices in the network are connected, except for the average path length. By contrast, for the average path length of a network, the smaller it is, the more closely the vertices are connected.</p>
</sec>
<sec id="s2-4">
<title>2.4 K-means clustering algorithm and elbow method</title>
<p>Suppose that data can be divided into <italic>K</italic> classes (<italic>K</italic> is known). The data are a set of <italic>n</italic> observations, that is, {<italic>x</italic>
<sub>1</sub>, &#x2026;, <italic>x</italic>
<sub>
<italic>n</italic>
</sub>}. The observations are divided into <italic>K</italic> non-intersecting sets {<italic>C</italic>
<sub>1</sub>, &#x2026;, <italic>C</italic>
<sub>
<italic>K</italic>
</sub>}, and all clusters are concatenated for the whole sample. <italic>i</italic> &#x2208; <italic>C</italic>
<sub>
<italic>k</italic>
</sub> means that the <italic>i</italic>-th observation <italic>x</italic>
<sub>
<italic>i</italic>
</sub> belongs to the <italic>k</italic>-th cluster. We want the &#x201c;in-group variation&#x201d; of each cluster to be as small as possible, and the mean or center position of cluster <italic>k</italic> is given by:<disp-formula id="e8">
<mml:math id="m27">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2261;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:math>
<label>(8)</label>
</disp-formula>where &#x7c;<italic>C</italic>
<sub>
<italic>k</italic>
</sub>&#x7c; denotes the number of observations in cluster <italic>k</italic>. There are <italic>K</italic> means in the sample, so we refer to the algorithm as <italic>K-means clustering</italic>. For an observation <italic>x</italic>
<sub>
<italic>i</italic>
</sub>(<italic>i</italic> &#x2208; <italic>C</italic>
<sub>
<italic>k</italic>
</sub>) in cluster <italic>k</italic>, the deviation (<italic>x</italic>
<sub>
<italic>i</italic>
</sub> &#x2212; <italic>c</italic>
<sub>
<italic>k</italic>
</sub>) from the center of the cluster is called the &#x201c;error.&#x201d; The sum of squares of all the errors in cluster <italic>k</italic> is the sum of squares error (SSE) of cluster <italic>k</italic>:<disp-formula id="e9">
<mml:math id="m28">
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2261;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:math>
<label>(9)</label>
</disp-formula>where &#x2016;<bold>x</bold>
<sub>
<italic>i</italic>
</sub> &#x2212; <bold>c</bold>
<sub>
<italic>k</italic>
</sub>&#x2016; is the Euclidean distance. The SSE of all clusters is summed to obtain the SSE of the full sample; the objective is to find a division <italic>C</italic>
<sub>1</sub>, &#x2026;, <italic>C</italic>
<sub>
<italic>K</italic>
</sub> for the set of sample subscripts 1, &#x2026;, <italic>n</italic> that minimizes the SSE of:<disp-formula id="e10">
<mml:math id="m29">
<mml:munder>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x2261;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>Because it is difficult to determine the global minimum solution to the problem, K-means clustering generally uses an iterative algorithm to find the local minimum solution. The results obtained will depend on the initial (random) cluster assignment of each observation. For this reason, it is important to run the algorithm multiple times from different random initial configurations. Another key issue in using K-means clustering is how the number of clusters <italic>K</italic> should be chosen.</p>
<p>The elbow method as a heuristic is a commonly used cluster analysis method, which helps determine the optimum number <italic>K</italic> of clusters. The method can improve the accuracy and reliability of cluster analysis. The basic principle of the elbow method is to determine the optimal number of clusters by calculating the sum of SSE for different numbers of clusters. When the number of clusters increases, the SSE decreases, but the rate of decrease slows. When the number of clusters increases to a certain point, the decrease in SSE slows rapidly, forming an elbow. This elbow is the key to the K-means elbow rule, which indicates the optimal number of clusters.</p>
</sec>
</sec>
<sec id="s3">
<title>3 TDA-based systemic financial risk analysis</title>
<sec id="s3-1">
<title>3.1 Construction of point clouds data and persistence landscapes</title>
<p>For each trading day <italic>t</italic>, we consider all stocks as points to construct the point cloud data and carry out TDA on the complexes, mainly based on the persistence landscapes. We consider the 0- and 1-dim barcodes and calculate the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms of the persistence landscapes both together. We first choose the length of the sliding window to obtain reasonable parameters for subsequent analysis. After comparison, the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms clearly exhibit better results with a sliding window <italic>&#x3c9;</italic> &#x3d; 100.</p>
<p>In <xref ref-type="fig" rid="F3">Figure 3</xref>, we present an overall description of 100 stocks and want to compare the general indicators with the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms to illustrate the significance of using the TDA measure. We plot the average of 100 stocks&#x2019; return series and the average of their standard deviations in <xref ref-type="fig" rid="F3">Figures 3A,B</xref>, respectively. The standard deviation, a common method for calculating volatility, measures the degree of volatility of a particular return series. Thus, the average standard deviation of the 100 stocks in <xref ref-type="fig" rid="F3">Figure 3B</xref> can approximately depict the state of our stock market and also capture the turbulence in the market (manifested as an increase in the standard deviation).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Common indicators and corresponding norms series of the 100 stocks over the full sample period. <bold>(A)</bold> Average daily log returns of 100 stocks. <bold>(B)</bold> Average standard deviations of 100 stocks. Our aim is to document the fluctuations of the market approximately. <bold>(C)</bold> L<sup>1</sup>- and L<sup>2</sup>-norms of persistent landscapes are calculated using both 0- and 1-dim barcodes; the period is from 2005/06/09 to 2022/03/31.</p>
</caption>
<graphic xlink:href="fphy-11-1253953-g003.tif"/>
</fig>
<p>The <italic>L</italic>
<sup>1</sup>-and <italic>L</italic>
<sup>2</sup>-norms of persistence landscapes of the point cloud data are depicted in <xref ref-type="fig" rid="F3">Figure 3C</xref>. In detail, the norms are based on the simplicial complex of high-dimensional point clouds, so the values of the norms are closely related to the distances and the structures of the point clouds. The smaller the norms correspond, the closer the distances; that is, the stronger the correlation between the stocks. We can observe the following phenomenon. Compared to the norms we obtained, the valleys of the norms are significantly less than the peaks of the standard deviation. This is because if the log return persists at a high level of volatility for a long period, its corresponding norms are small. Conversely, if the log return is very high in volatility at a time point or sits at a high level of volatility for a short period, its corresponding norms are at normal levels. This illustrates that the norms calculated from the topological perspective are very robust to outliers.</p>
<p>Obviously, the <italic>L</italic>
<sup>1</sup>-and <italic>L</italic>
<sup>2</sup>-norms were low and recovering from 2005, influenced by the SARS epidemic in 2003. Beginning at the end of October 2007, the return fluctuated rapidly and sharply, and the norms hit bottom in 2008 owing to the U.S. Subprime Lending Crisis along with the Wenchuan earthquake in China. The standard deviation continued to be at a high level for a long period of time and the <italic>L</italic>
<sup>1</sup>-and <italic>L</italic>
<sup>2</sup>-norms declined significantly. The link among the stocks became closer, and the situation was severe. From December 2014 to May 2015, there was a fast-rising bull market in China&#x2019;s stock market, during which the Shanghai and Shenzhen stock indices rose in turn (the Shanghai Composite Index and Shenzhen Stock Index accumulated 53.17% and 57.13%, respectively). Subsequently, the China Securities Regulatory Commission began to rectify illegal allocation and forced deleveraging in the short term, leading to a continuous decline in the A-share market. From June to September, the Shanghai Composite Index cumulatively declined &#x2212;42.26% and the Shenzhen Composite Index declined &#x2212;48.74%. The standard deviation rose to a record peak. This, in turn, led to successive plunges in various stock market-style assets. Hence, China&#x2019;s 2015 stock market crash led to an unstable economic situation, and the norms rapidly fell to record valleys. The connections among stocks became increasingly close during the crisis period. Since 2018, China&#x2019;s stock market has no longer been stable and flat due to the trade war between China and the U.S. and the COVID-19 pandemic. In general, when major events occur, the average standard deviation rises gradually. <italic>L</italic>
<sup>1</sup>-and <italic>L</italic>
<sup>2</sup>-norms exhibit obvious drops and are at a low level, indicating that the structure of the data changes and the connections among stocks becomes close. Whenever there is a sudden structural change, our persistence landscapes change. This leads to changes in <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms. This is especially evident when the structure shifts to tighter situations. The use of persistence landscapes and <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms as barometers of shocks enables timely information to be detected.</p>
</sec>
<sec id="s3-2">
<title>3.2 Critical dates detection and analysis</title>
<p>We want to extract the feature of the returns using the aforementioned data processing method. Guo et al. [<xref ref-type="bibr" rid="B20">20</xref>] used TDA and detected the early warning signal of the 2008 crisis in the first half of 2007. The TDA model provides a better definition of market signals, especially during financial meltdown cycles [<xref ref-type="bibr" rid="B21">21</xref>]. The trend of the <italic>L</italic>
<sup>1</sup>-norms is similar to that of <italic>L</italic>
<sup>2</sup>-norms, but the <italic>L</italic>
<sup>1</sup>-norms change more frequently. We predict that using the <italic>L</italic>
<sup>1</sup>-norms can detect more critical dates than the <italic>L</italic>
<sup>2</sup>-norms with the same other parameters and validate it through multiple tests.</p>
<p>We built a system to detect dates with sharp changes in point cloud data by referring to Guo et al. [<xref ref-type="bibr" rid="B20">20</xref>]. For every trading day&#xa0;<italic>t</italic>, we calculated and denoted by <italic>N</italic>(<italic>t</italic>) the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms of the persistence landscape of the complex (we calculated the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms of trading day&#xa0;<italic>t</italic> based on the sliding window from the point cloud data of trading day&#xa0;<italic>t</italic> &#x2212; <italic>&#x3c9;</italic> &#x2b; 1 to trading day&#xa0;<italic>t</italic>), and by <italic>A</italic>(<italic>t</italic>) the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms&#x2019; mean value of <italic>m</italic> consecutive trading days before day&#xa0;<italic>t</italic>. If the following two conditions are satisfied, we take trading day&#xa0;<italic>t</italic> as the critical date for a financial crisis or risk: i) The ratio <italic>N</italic>(<italic>t</italic>)/<italic>N</italic>(<italic>t</italic> &#x2212; 1) &#x2264; <italic>&#x3b1;</italic> for a fixed <italic>&#x3b1;</italic> &#x3c; 1. ii) The ratio <italic>N</italic>(<italic>t</italic>)/<italic>A</italic>(<italic>t</italic>) &#x2264; <italic>&#x3b2;</italic> for a fixed <italic>&#x3b2;</italic> &#x3c; 1. For simplicity, let <italic>m</italic> &#x3d; 100.</p>
<p>The first condition indicates the presence of risk on trading day&#xa0;<italic>t</italic>, whereas the second ensures that it is a true crisis-critical date. It is possible to identify multiple dates, which indicate significant stock market shocks, but some may be only brief declines caused by coincidental events that are not representative. Under the <italic>L</italic>
<sup>1</sup>-and <italic>L</italic>
<sup>2</sup>-norms, some of the critical dates detected for the parameters at different values are list in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Critical dates detected on different values of parameters.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Parameters</th>
<th align="center">Critical dates</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">
<italic>L</italic>
<sup>1</sup>-norm <italic>&#x3b1;</italic> &#x3d; 0.9, <italic>&#x3b2;</italic> &#x3d; 0.78</td>
<td align="center">2007/03/08,2008/04/02,2008/04/07,2008/04/28,2008/05/20,2008/06/10,2008/06/18,2015/08/20,2015/08/28,2015/11/02,2018/11/07,2018/12/03,2020/02/03,2020/04/01</td>
</tr>
<tr>
<td align="center">
<italic>L</italic>
<sup>1</sup>-norm <italic>&#x3b1;</italic> &#x3d; 0.88, <italic>&#x3b2;</italic> &#x3d; 0.8</td>
<td align="center">2008/04/02,2008/06/10,2018/12/03,2019/01/07,2020/02/03,2020/04/01,2022/03/29</td>
</tr>
<tr>
<td align="center">
<italic>L</italic>
<sup>2</sup>-norm <italic>&#x3b1;</italic> &#x3d; 0.93, <italic>&#x3b2;</italic> &#x3d; 0.86</td>
<td align="center">2008/04/02,2008/06/10,2018/12/03,2020/02/03,2020/04/01,2022/03/29</td>
</tr>
<tr>
<td align="center">
<italic>L</italic>
<sup>2</sup>-norm <italic>&#x3b1;</italic> &#x3d; 0.95, <italic>&#x3b2;</italic> &#x3d; 0.85</td>
<td align="center">2007/03/08,2008/02/26,2008/04/02,2008/04/07,2008/04/28,2008/05/20,2008/06/10,2008/06/18,2015/06/30,2015/0727,2015/08/20,2015/08/28,2015/11/02,2015/11/05,2018/11/07,2018/12/03,2020/02/03,2020/02/05,2020/04/01</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From <xref ref-type="table" rid="T1">Table 1</xref>, the highest number of critical dates were detected for 2007&#x2013;2008, with 2008/04/02 and 2008/06/10 appearing most frequently, followed by critical dates detected under different combinations of parameters for 2015 (2015/08/20, 2015/08/28, and 2015/11/02). The critical date in 2018 is 2018/12/03, those in 2020 are 2020/02/03 and 2020/04/01, and for 2022 is 2022/03/29. Considering that some events last longer and even had multiple rounds of shocks, we used the first identified dates in the critical years as representative critical dates for comparability: 2008/04/02, 2015/08/20, 2018/12/03, 2020/02/03,4 and 2022/03/29.</p>
<p>We use the same method to obtain the log return of the CSI 300 Index. Using the financial crisis-critical date detection system for its log returns, there are many sudden change dates, and it does not focus on identifying the real crisis-critical dates. In contrast, the norm series can better focus on and capture the critical dates of a real crisis, so it will be more effective to use this to perform region transformation in the future.</p>
<p>We analyze the stocks&#x2019; relevance for all critical dates and the 50th days before. To save space, <xref ref-type="fig" rid="F4">Figure 4A&#x2013;D</xref> present only the persistent landscapes composed of point clouds before and at the time of two major events, depicting the evolution of data shapes before and during each critical date. Two endpoints of the intersection of each layer with the horizontal axis correspond to the left and right endpoints of each interval of the 1-dim barcode, respectively. The highest layer corresponds to the longest interval of the barcode. The comparison indicates different degrees of change in persistent landscapes before and after the first outbreak of the two events. Those in <xref ref-type="fig" rid="F4">Figure 4B</xref> tend to become smaller in most triangles compared with those in <xref ref-type="fig" rid="F4">Figure 4A</xref>; that is, the norms are significantly lower. Compared to <xref ref-type="fig" rid="F4">Figure 4C</xref>, the persistent landscapes in <xref ref-type="fig" rid="F4">Figure 4D</xref> exhibit a significant decrease in the highest layer, and the right side of the triangle is no longer concentrated in 0.6&#x2013;1.1 but is more scattered in 0.4&#x2013;1, thus limiting the total area of the inner layers to decrease. This reflects the decrease in the norm. Moreover, pre-crisis to crisis norms decrease more. We obtain good evidence that the stock market links strengthen when a crisis occurs.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Persistent landscapes on 2008/01/16, 2008/04/02, 2019/11/14, and 2020/02/03 from <bold>(A&#x2013;D)</bold>, respectively. The figures contain many <inline-formula id="inf20">
<mml:math id="m30">
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> functions (i.e., triangles). The triangles with left vertices starting from 0 are calculated from 0-dim barcodes, and the others are from 1-dim barcodes. <bold>(A)</bold> Persistance landscapes on 2008/01/16. There are 132 f<sub>(aj, bj)</sub> functions. <bold>(B)</bold> Persistance landscapes on 2008/04/02. There are 94 f<sub>(aj, bj)</sub> functions. <bold>(C)</bold> Persistance landscapes on 2019/11/14. There are 97 f<sub>(aj, bj)</sub> functions. <bold>(D)</bold> Persistance landscapes on 2020/02/03. There are 81 f<sub>(aj, bj)</sub> functions.</p>
</caption>
<graphic xlink:href="fphy-11-1253953-g004.tif"/>
</fig>
<p>We plotted threshold networks for each of the five pairs of critical dates identified. To obtain a reasonably uniform threshold <italic>&#x3b8;</italic>, we calculated and compared the degrees of networks of the 10 aforementioned dates, and separately differed the degrees of networks of the critical dates from their corresponding previous 50th days. When <italic>&#x3b8;</italic> &#x3d; 1.1, the average difference of the network average degrees for the five pairs of critical dates reaches the maximum level. The comparison of the topological indicators of the return threshold network is presented in <xref ref-type="table" rid="T2">Table 2</xref>. Our stock market moved the most during the financial crisis triggered by the U.S. Subprime Lending Crisis in 2008 and the stock market crash in 2015, followed by the trade war between China and the U.S. in 2018 and the COVID-19 pandemic in 2020 and 2022. Before and after the U.S. Subprime Lending Crisis in 2008 and the stock market crash in 2015, the degree of the network increased from 26.58 to 41.68 to 87.52 and 82.88, the network density increased from 0.268 to 0.421 to 0.884 and 0.837, the clustering coefficient of the network changed from 0.61 to 0.749 to 0.945 and 0.931, and the average path length of the network from 1.85 to 1.602 to 1.098 and 1.146, respectively. Hence, the connection between stocks became stronger. The return threshold networks for 2015/06/10 and 2015/08/20 are presented in <xref ref-type="fig" rid="F5">Figure 5</xref>. The number of connected edges in the network increases significantly before and after the event, indicating that the stock market is closely linked under major event shocks. This result is consistent with previous findings.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>The comparison of the topological indicators of the return threshold network.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Date</th>
<th align="center">Degree</th>
<th align="center">Density</th>
<th align="center">Clustering coefficient</th>
<th align="center">Average path length</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">2008/01/16</td>
<td align="center">26.58</td>
<td align="center">0.268</td>
<td align="center">0.61</td>
<td align="center">1.85</td>
</tr>
<tr>
<td align="center">2008/04/02</td>
<td align="center">87.52</td>
<td align="center">0.884</td>
<td align="center">0.945</td>
<td align="center">1.098</td>
</tr>
<tr>
<td align="center">2015/06/10</td>
<td align="center">41.68</td>
<td align="center">0.421</td>
<td align="center">0.749</td>
<td align="center">1.602</td>
</tr>
<tr>
<td align="center">2015/08/20</td>
<td align="center">82.88</td>
<td align="center">0.837</td>
<td align="center">0.931</td>
<td align="center">1.146</td>
</tr>
<tr>
<td align="center">2018/09/14</td>
<td align="center">45.96</td>
<td align="center">0.464</td>
<td align="center">0.800</td>
<td align="center">1.580</td>
</tr>
<tr>
<td align="center">2018/12/03</td>
<td align="center">75.90</td>
<td align="center">0.767</td>
<td align="center">0.891</td>
<td align="center">1.234</td>
</tr>
<tr>
<td align="center">2019/11/14</td>
<td align="center">35.12</td>
<td align="center">0.355</td>
<td align="center">0.748</td>
<td align="center">1.777</td>
</tr>
<tr>
<td align="center">2020/02/03</td>
<td align="center">60.76</td>
<td align="center">0.614</td>
<td align="center">0.84</td>
<td align="center">1.447</td>
</tr>
<tr>
<td align="center">2022/01/11</td>
<td align="center">6.32</td>
<td align="center">0.064</td>
<td align="center">0.550</td>
<td align="center">3.541</td>
</tr>
<tr>
<td align="center">2022/03/29</td>
<td align="center">27.64</td>
<td align="center">0.279</td>
<td align="center">0.679</td>
<td align="center">1.931</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Comparison of the return networks for 2015/06/10 and 2015/08/20. For example, in 2015, we find a significant increase in the number of connected edges in the network when an event occurs. <bold>(A)</bold> Return network on 2015/06/10. <bold>(B)</bold> Return network on 2015/08/20.</p>
</caption>
<graphic xlink:href="fphy-11-1253953-g005.tif"/>
</fig>
<p>The correlation coefficients and corresponding frequent dynamic distribution, calculated from the daily log returns with a rolling window of 100 trading days, are illustrated in <xref ref-type="fig" rid="F6">Figure 6</xref>. The correlation coefficients among stocks at normal times are mostly distributed in the interval of &#x2212;0.25 to 0.50, and the corresponding probability distributions are between 0 and 0.1, which are approximately normal. This indicates that the correlations among stocks are not significant, and the clustering effect is not obvious. However, the peaks of the cross-sectional graphs of the frequency distribution of correlation coefficients in 2008 and 2015 are higher. Therefore, we think the correlation coefficients among stocks generally tend to rise and cluster during a sudden event. The changes in 2018 and 2020 are smaller than the previous two systemic risks, confirming that differences exist in the degree of impact of each major event on China. The following section analyzes the evolution of heterogeneity in the structure of the stock market under different events using the higher-order characteristics of stocks (i.e., neighborhoods).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Correlation coefficients and corresponding frequent dynamic distribution of 100 stocks. The lateral axis represents time, the longitudinal axis represents the correlation coefficient values, and the vertical axis represents the frequency of the correlation coefficients appearing in a certain interval at a specific time point.</p>
</caption>
<graphic xlink:href="fphy-11-1253953-g006.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4 Comparison and mechanisms of neighborhood norms before and during the events</title>
<sec id="s4-1">
<title>4.1 Overall neighborhood norms and their applications</title>
<p>After obtaining the distance between stocks in the overall sample interval, we first select the 30 stocks with the closest distance in the neighborhood for each stock to construct new 31 &#xd7; 31 distance matrices <italic>D</italic>
<sub>
<italic>i</italic>,<italic>total</italic>
</sub>. The subscript <italic>i</italic> &#x3d; 1, &#x2026;, 100 represents one stock, and <italic>total</italic> represents the distance matrix we used for the full sample interval. Second, we consider 31 stocks as points to construct the point cloud data and conduct TDA on the complexes, mainly based on the persistence landscapes. Finally, we use the same method to calculate the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms of the persistence landscapes, called <italic>neighborhood norms</italic>. To explore the dynamic evolution of the stock market structure before and during the event, after determining the number of clusters using the elbow method, we performed a cluster analysis using the K-means on the neighborhood norms. The elbow plot of the neighborhood norms during the full-time interval is illustrated in <xref ref-type="fig" rid="F7">Figure 7</xref>. When the number of clusters gradually increases from one to four, the SSE decreases in a decreasing trend with a larger magnitude, and when the number of clusters continues to increase from four to five or more, the decrease in SSE becomes small. Using the elbow method, we selected the elbow corner of SSE in figure <italic>K</italic> &#x3d; 4 for the number of clusters. Therefore, we obtained the <italic>L</italic>
<sup>
<italic>p</italic>
</sup>-norms contained in each category, denoted <inline-formula id="inf21">
<mml:math id="m31">
<mml:msubsup>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula>-norms. The subscript <italic>j</italic> &#x3d; 1, &#x2026;, 4 denotes the category of clustering from the smallest to the largest based on the norms. To quantify the changes in the norms, let <italic>p</italic> &#x3d; 1, that is, the <italic>L</italic>
<sup>1</sup>-norms are considered as an example in the following.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Elbow plot of the neighborhood norms for the overall sample interval. The lateral axis represents the number of clusters <italic>K</italic>, and the vertical axis represents the sum of squared errors (SSE) of the full sample. The bend in the line, the point where the SSE descent becomes slower, is evident.</p>
</caption>
<graphic xlink:href="fphy-11-1253953-g007.tif"/>
</fig>
<p>The industry segmentation is determined by the main business of the listed company, which in turn influences the price through fundamental factors; therefore, so the volatility of stock prices in the same industry is more similar. Therefore, according to the latest SIC, we conduct in-depth research on 100 stocks by industry. Owing to the large number of industries involved, we represent them by industry codes<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref>. We can draw the following conclusions from <xref ref-type="table" rid="T3">Table 3</xref>. In general, the distribution of stocks in most industries is more uniform and scattered, and the owning industries of several mass organizations, including stock structures, are rich. This is due to the influence of financial market psychology, such as the herd effect and market expectations, which are benign and reflect the normal characteristics of China&#x2019;s stock markets. The first category is closely related to other stocks; it basically covers the representative stocks in each industry, such as wholesale and retail. Financial sector stocks are distributed in the second and third categories, which indicates that most are at the midstream level with surrounding connections because these play a role in stabilizing the stock markets. The manufacturing industry and so on dominate in all categories.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The categorization of neighborhood norms and industry distribution of the full sample.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Each category L<sub>ij</sub>
<sup>p</sup>
</th>
<th align="center">Industry codes (From large to small in proportion)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Category 1</td>
<td align="center">C, A, B, F, I, D, G, H, K</td>
</tr>
<tr>
<td align="center">Category 2</td>
<td align="center">J, C, G, N, K, D, B, F, I</td>
</tr>
<tr>
<td align="center">Category 3</td>
<td align="center">C,J, B, G, K, D, I, A</td>
</tr>
<tr>
<td align="center">Category 4</td>
<td align="center">C, D</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2">
<title>4.2 Stock correlation changes around critical dates and their mechanisms</title>
<p>Whether it is a complementary or a collaborative relationship, there are certain connections between industries. We want to provide a more accurate characterization of how the industry is changing and influenced by major events. Therefore, we use each of the aforementioned critical dates and the 50th day before them as the scope for the industry analysis. We then apply TDA to calculate the neighborhood norms separately for all stocks on a day <italic>t</italic> and use the <italic>K</italic>-means to cluster the norms, respectively. According to each elbow graph, <italic>K</italic> &#x3d; 4 is always the optimal cluster number, and finally obtain the neighborhood norms <inline-formula id="inf22">
<mml:math id="m32">
<mml:msubsup>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> of each category under critical dates. The subscript <italic>t</italic> &#x3d; 1, &#x2026;, 10 denotes the dates in order.</p>
<p>We present the changes in the neighborhood norms for the 100 stocks for each pair of dates in <xref ref-type="fig" rid="F8">Figure 8</xref>. Although the <italic>L</italic>
<sup>1</sup>-norms before and during events are different, they generally display a synchronous trend, that is, the neighborhood norms before events are always greater than those during events. This illustrates that when the event occurs, each stock is influenced by other stocks in all industries. It is more consistent with the norms of the overall sample interval, which indicates that there is risk linkage among stocks in various industries, and also demonstrates the stability and applicability of the norms and categorization.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Comparison of neighborhood norms before and during events from <bold>(A&#x2013;E)</bold>, respectively. <bold>(A)</bold> Neighborhood norms of 100 stocks on 2008/01/16 and 2008/04/02. <bold>(B)</bold> Neighborhood norms of 100 stocks on 2015/06/10 and 2015/08/20. <bold>(C)</bold> Neighborhood norms of 100 stocks on 2018/09/14 and 2018/12/03. <bold>(D)</bold> Neighborhood norms of 100 stocks on 2019/11/14 and 2020/02/03. <bold>(E)</bold> Neighborhood norms of 100 stocks on 2022/01/11 and 2022/03/29.</p>
</caption>
<graphic xlink:href="fphy-11-1253953-g008.tif"/>
</fig>
<p>We use K-means to cluster the neighborhood norms before and during events (<xref ref-type="table" rid="T4">Tables 4</xref>, <xref ref-type="table" rid="T5">5</xref>, respectively). There are some industries where stocks are distributed and dispersed, both before and during events. For example, the manufacturing industry is in almost every category. This illustrates that China&#x2019;s stocks in the manufacturing industry are in a relatively stable state before and during events. There are also some industries whose stocks have relatively significant changes in their distribution. For example, the mining and real estate industries have large changes in their categories in response to different shocks; that is, there is an increase in correlation of different magnitudes.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Comparison and industry distribution of neighborhood norms before events.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Date Category</th>
<th align="center">1</th>
<th align="center">2</th>
<th align="center">3</th>
<th align="center">4</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">2008/01/16</td>
<td align="center">C,G,I,D,A</td>
<td align="center">C,K,B,D,J,N,A,G,F</td>
<td align="center">J,C,G,B,D,A,K,H,I,F</td>
<td align="center">C,J,F,G,I,D,B,K</td>
</tr>
<tr>
<td align="center">2015/06/10</td>
<td align="center">J,D,C,B,K,G,N</td>
<td align="center">C,G,F,A,B,D,K,N</td>
<td align="center">C,I,B,G,K,A</td>
<td align="center">I,H,C,J</td>
</tr>
<tr>
<td align="center">2018/09/14</td>
<td align="center">C,J,B,K,I,G,D,N,F,H</td>
<td align="center">C,G,I,K,B,F,D,A</td>
<td align="center">C,A,G,J,D</td>
<td align="center">C,G</td>
</tr>
<tr>
<td align="center">2019/11/14</td>
<td align="center">C,J,D,K,B,F,G,H,I</td>
<td align="center">C,G,J,I,K,N,A,F</td>
<td align="center">C,G,B,A</td>
<td align="center">C,D,A,G,B,J</td>
</tr>
<tr>
<td align="center">2022/01/11</td>
<td align="center">J,C,K,B,N,D</td>
<td align="center">G,C,D,J,B,A,K,H,I,F</td>
<td align="center">C,B,G,K,F,N,A,I,J</td>
<td align="center">C,I,D,G,A,B,F</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Comparison and industry distribution of neighborhood norms during events.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Date Category</th>
<th align="center">1</th>
<th align="center">2</th>
<th align="center">3</th>
<th align="center">4</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">2008/04/02</td>
<td align="center">C,G,I,D,F,B</td>
<td align="center">C,A,N,I,D,B,J,K,H,G</td>
<td align="center">J,K,C,G,A,B,D</td>
<td align="center">J,C,F,D</td>
</tr>
<tr>
<td align="center">2015/08/20</td>
<td align="center">C,G,K,F,D,B,A,I,J</td>
<td align="center">C,G,A,D,B,N,F,I</td>
<td align="center">J,C,K,D,I,B,G,H</td>
<td align="center">J</td>
</tr>
<tr>
<td align="center">2018/12/03</td>
<td align="center">C,J,B,D,K,F,H,G,A,I</td>
<td align="center">G,C,J,I,N,F,D,A,K,B</td>
<td align="center">C,J,I,K,B,A</td>
<td align="center">C,G,A,B,K,D</td>
</tr>
<tr>
<td align="center">2020/02/03</td>
<td align="center">C,J,K,F,D,B,G,N,H,A</td>
<td align="center">G,C,J,D,I,A,B,K</td>
<td align="center">C,I,B,A,J</td>
<td align="center">C</td>
</tr>
<tr>
<td align="center">2022/03/29</td>
<td align="center">J,C,K,B,D,G</td>
<td align="center">G,D,C,I,J,B,N,H,K</td>
<td align="center">C,F,I,B,A,N</td>
<td align="center">C,G,A,J</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In 2008, the Subprime Lending Crisis caused significant changes in the stock markets. The stock correlation of building materials and furniture industries strengthened, indicating that the crisis had a very severe impact on export-dependent industries. There is also a significant increase in stock correlations in the mining industry. The crash in China&#x2019;s stock markets in 2015 led to a significant increase in stock correlations in the farming, forestry, animal husbandry, and fishery industry and wholesale and retail trade, and so on. In 2018, industries were affected by the trade war between China and the U.S.; the changes in correlations are more consistent, and a few stocks have smaller changes. The COVID-19 pandemic in 2020 has dealt an irretrievable blow to the global real economy. Electrical, heat, gas, water production and supply industry, and so on. are mainly located in the first two categories, and wholesale and retail trade are mainly located in the first category; the correlation of all of them has increased significantly. In 2022, the COVID-19 outbreak was concentrated in Shanghai. The correlation between stocks in the electrician, heat, gas, and water production and supply industry and the real estate industry all increased significantly.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>We use TDA and K-means to examine the variation in China&#x2019;s stocks under major events. The main conclusions are as follows. First, according to the TDA and norms, when the financial crisis occurred, stocks were closely linked. During a major public event, the norms are obviously at a lower level, which reveals the strong correlations among the stocks. The norms can reasonably and effectively depict the volatility of the stock markets and accurately identify critical information. Second, the threshold network and its topological measures indicate that, with the arrival of the financial crisis, the correlation among sample stocks became increasingly close in the short term, but the heterogeneity of systemic risks caused by major events led to different changes in the degree of correlation between stock returns. Third, before an event, stocks in most industries were evenly distributed, and those in each community structure belonged to various industries. During an event, the stocks of each industry had a greater impact on the other stocks. Consistency and agglomeration reveal a risk linkage between the stocks of each industry. The impact of an event is systematic, with rapid risk and complex transmission paths. From another perspective, different industries were affected by different events with different direct and indirect impacts. When we examine some asset-specific issues or different countries&#x2019; stock markets, their persistence landscapes and norms change with the structure of stock prices, thus identifying the critical dates. Our methodology is generalizable and can be extended to other countries, where information about changes in other countries is closely linked to major events in their countries, and also to identify mutation dates and thus investigate market changes.</p>
<p>Based on our results, we propose the following suggestions. First, strengthen the target of financial coordination regulation. Furthermore, various industries should be divided into multiple levels of policy assistance and supervision according to the degree of correlation and the strength of impact to improve the efficiency of resource allocation in the vulnerable period of the financial markets. Second, the double effects of supply and demand have a profound effect on each industry, particularly when major events occur. The interdependence of each industry&#x2019;s economy is evident, and a chain reaction is in effect. Therefore, we must provide risk protection for enterprises while simultaneously improving the efficiency of capital use and the fiscal deficit ratio to ensure sustainable financial development. We also believe that our methods have a guiding role in capturing market variability. In addition, as the world&#x2019;s second-largest economy, frequent changes in China&#x2019;s stock market are bound to affect other economic markets, even the global economy. We aim to conduct a deeper study on the stock markets, financial markets, and even the macroeconomy by adding more stock data, indicators, or statistical information and continue to expand the scope of research to explore the changes in global stock markets, financial markets, and even the macroeconomy. In future research, the use of <italic>L</italic>
<sup>
<italic>p</italic>
</sup> distance metrics matrix [<xref ref-type="bibr" rid="B27">27</xref>] can be attempted to classify and evaluate crises and risks based on TDA and machine learning to realize risk warning more effectively.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement </title>
<p>The raw data supporting the conclusion of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>HG: Conceptualization, Data curation, Formal Analysis, Methodology, Project administration, Software, Supervision, Validation, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing. ZM: Data curation, Formal Analysis, Software, Visualization, Writing&#x2013;original draft. BX: Data curation, Formal Analysis, Validation, Visualization, Writing&#x2013;review and editing. All authors listed have made a substantial, direct, and intellectual contribution to the work and approved it for publication.</p>
</sec>
<sec id="s8">
<title>Funding </title>
<p>This research was funded by the National Natural Science Foundation of China-Shandong Joint Fund (No. U1806203), the National Social Science Foundation of China (No. 21BTJ072), the Shandong Key R <italic>&#x26;</italic> D (major scientific and technological innovation) Project (No. 2021CXGC010108), and the Shandong Undergraduate Teaching Reform Research Project (No. 2022096).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest </title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>Farming, forestry, animal husbandry, and fishery industry-A, mining industry-B, manufacturing industry-C, electrical, heat, gas, and water production and supply-D, wholesale and retail trade-F, transportation, warehousing, and postal services-G, hotel and catering sectors-H, information transmission, software, and information technology services-I, finance-J, real estate-K, water conservancy, environment, and public facilities management industry-N.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>James</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Menzies</surname>
<given-names>M</given-names>
</name>
</person-group>. <article-title>Association between covid-19 cases and international equity indices</article-title>. <source>Physica D Nonlinear Phenomena</source> (<year>2021</year>) <volume>417</volume>:<fpage>132809</fpage>. <pub-id pub-id-type="doi">10.1016/j.physd.2020.132809</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Omay</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Iren</surname>
<given-names>P</given-names>
</name>
</person-group>. <article-title>Behavior of foreign investors in the malaysian stock market in times of crisis: a nonlinear approach</article-title>. <source>J Asian Econ</source> (<year>2019</year>) <volume>60</volume>:<fpage>85</fpage>&#x2013;<lpage>100</lpage>. <pub-id pub-id-type="doi">10.1016/j.asieco.2018.11.002</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>Analysis of systemic risk from the perspective of complex networks: overview and outlook</article-title>. <source>Control Theor Appl</source> (<year>2022</year>) <volume>39</volume>:<fpage>2202</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.7641/CTA.2021.10267</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Diebold</surname>
<given-names>FX</given-names>
</name>
<name>
<surname>Y&#x131;lmaz</surname>
<given-names>K</given-names>
</name>
</person-group>. <article-title>On the network topology of variance decompositions: measuring the connectedness of financial firms</article-title>. <source>J Econom</source> (<year>2014</year>) <volume>182</volume>:<fpage>119</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1016/j.jeconom.2014.04.012</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gong</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W</given-names>
</name>
</person-group>. <article-title>Research on systemic risk measurement and spillover effect of financial institutions in China</article-title>. <source>Manage World</source> (<year>2020</year>) <volume>36</volume>:<fpage>65</fpage>&#x2013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.3969/j.issn.1002-5502.2020.08.007</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adrian</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Brunnermeier</surname>
<given-names>MK</given-names>
</name>
</person-group>. <source>Covar <italic>Am Econ Rev</italic>
</source> (<year>2016</year>) <volume>106</volume>:<fpage>1705</fpage>&#x2013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.1257/aer.20120555</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jia</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>C</given-names>
</name>
</person-group>. <article-title>Var model in the application of the stock market risk analysis and empirical analysis</article-title>. <source>Chin J Manage Sci</source> (<year>2014</year>) <volume>22</volume>:<fpage>336</fpage>&#x2013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.16381/j.cnki.issn1003-207x.2014.s1.061</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Acharya</surname>
<given-names>VV</given-names>
</name>
<name>
<surname>Pedersen</surname>
<given-names>LH</given-names>
</name>
<name>
<surname>Philippon</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Richardson</surname>
<given-names>M</given-names>
</name>
</person-group>. <article-title>Measuring systemic risk</article-title>. <source>Rev Financial Stud</source> (<year>2016</year>) <volume>30</volume>:<fpage>2</fpage>&#x2013;<lpage>47</lpage>. <pub-id pub-id-type="doi">10.1093/rfs/hhw088</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brownlees</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Engle</surname>
<given-names>RF</given-names>
</name>
</person-group>. <article-title>Srisk: a conditional capital shortfall measure of systemic risk</article-title>. <source>Rev Financial Stud</source> (<year>2016</year>) <volume>30</volume>:<fpage>48</fpage>&#x2013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1093/rfs/hhw060</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Identification of global systemic risk contagion path based on multiple motivations of international capital flow</article-title>. <source>Stat Res</source> (<year>2021</year>) <volume>38</volume>:<fpage>42</fpage>&#x2013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.19343/j.cnki.11-1302/c.2021.12.004</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>The co-movement and risk spillover effects of Chinese and foreign stock markets under the impact of sudden events</article-title>. <source>J Quantitative Econ</source> (<year>2022</year>) <volume>13</volume>:<fpage>15</fpage>&#x2013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.16699/b.cnki.jqe.2022.01.006</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>F</given-names>
</name>
</person-group>. <article-title>A study of stock market risk spillover effects: an analysis based on evt-copula-covar model</article-title>. <source>The J World Economy</source> (<year>2011</year>) <volume>339</volume>:<fpage>145</fpage>&#x2013;<lpage>59</lpage>. <pub-id pub-id-type="doi">10.19985/j.cnki.cassjwe.2011.11.009</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>F</given-names>
</name>
</person-group>. <article-title>Empirical analysis of relevance of stock indicators based on complex network theory</article-title>. <source>Chin J Manage Sci</source> (<year>2014</year>) <volume>22</volume>:<fpage>85</fpage>&#x2013;<lpage>92</lpage>. <pub-id pub-id-type="doi">10.16381/j.cnki.issn1003-207x.2014.12.012</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>An</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X</given-names>
</name>
</person-group>. <article-title>An empirical study of risk measurement in China&#x2019;s stock market - taking the sse index as an example</article-title>. <source>Stat Manage</source> (<year>2015</year>) <volume>213</volume>:<fpage>51</fpage>&#x2013;<lpage>2</lpage>. <pub-id pub-id-type="doi">10.3969/j.issn.1674-537X.2015.04.24</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robert</surname>
<given-names>G</given-names>
</name>
</person-group>. <article-title>Barcodes: the persistent topology of data</article-title>. <source>Bull Am Math Soc</source> (<year>2008</year>) <volume>45</volume>:<fpage>61</fpage>&#x2013;<lpage>75</lpage>. <pub-id pub-id-type="doi">10.1090/S0273-0979-07-01191-3</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carlsson</surname>
<given-names>G</given-names>
</name>
</person-group>. <article-title>Topology and data</article-title>. <source>Bull Am Math Soc</source> (<year>2009</year>) <volume>46</volume>:<fpage>255</fpage>&#x2013;<lpage>308</lpage>. <pub-id pub-id-type="doi">10.1090/s0273-0979-09-01249-x</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bubenik</surname>
<given-names>P</given-names>
</name>
</person-group>. <article-title>Statistical topology data analysis using persistence landscapes</article-title>. <source>J Machine Learn Res</source> (<year>2015</year>) <volume>16</volume>:<fpage>77</fpage>&#x2013;<lpage>102</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1207.6437</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bubenik</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Dlotko</surname>
<given-names>P</given-names>
</name>
</person-group>. <article-title>A persistence landscapes toolbox for topological statistics</article-title>. <source>J Symbolic Comput</source> (<year>2017</year>) <volume>78</volume>:<fpage>91</fpage>&#x2013;<lpage>114</lpage>. <pub-id pub-id-type="doi">10.1016/j.jsc.2016.03.009</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gidea</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Katz</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Topological data analysis of financial time series: landscapes of crashes</article-title>. <source>Physica A: Stat Mech its Appl</source> (<year>2018</year>) <volume>491</volume>:<fpage>820</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1016/j.physa.2017.09.028</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>S</given-names>
</name>
<name>
<surname>An</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X</given-names>
</name>
</person-group>. <article-title>Empirical study of financial crises based on topological data analysis</article-title>. <source>Physica A: Stat Mech its Appl</source> (<year>2020</year>) <volume>558</volume>:<fpage>124956</fpage>. <pub-id pub-id-type="doi">10.1016/j.physa.2020.124956</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goel</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Pasricha</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Mehra</surname>
<given-names>A</given-names>
</name>
</person-group>. <article-title>Topological data analysis in investment decisions</article-title>. <source>Expert Syst Appl</source> (<year>2020</year>) <volume>147</volume>:<fpage>113222</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2020.113222</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yen</surname>
<given-names>TW</given-names>
</name>
<name>
<surname>Cheong</surname>
<given-names>SA</given-names>
</name>
</person-group>. <article-title>Using topological data analysis (tda) and persistent homology to analyze the stock markets in singapore and taiwan</article-title>. <source>Front Phys</source> (<year>2021</year>) <volume>9</volume>:<fpage>572216</fpage>. <pub-id pub-id-type="doi">10.3389/fphy.2021.572216</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Munkres</surname>
<given-names>JR</given-names>
</name>
</person-group>. <source>Elements of algebraic topology</source>. <publisher-loc>California</publisher-loc>: <publisher-name>Addison Wesley</publisher-name> (<year>1984</year>). <pub-id pub-id-type="doi">10.1201/9780429493911</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Collins</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Zomorodian</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Carlsson</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Guibas</surname>
<given-names>LJ</given-names>
</name>
</person-group>. <article-title>A barcode shape descriptor for curve point cloud data</article-title>. <source>Comput Graphics</source> (<year>2004</year>) <volume>28</volume>:<fpage>881</fpage>&#x2013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1016/j.cag.2004.08.015</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cohen-Steiner</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Edelsbrunner</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Harer</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>Stability of persistence diagrams</article-title>. <source>Discrete Comput Geometry</source> (<year>2007</year>) <volume>37</volume>:<fpage>103</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1007/s00454-006-1276-5</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Costa</surname>
<given-names>LDF</given-names>
</name>
<name>
<surname>Rodrigues</surname>
<given-names>FA</given-names>
</name>
<name>
<surname>Travieso</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Boas</surname>
<given-names>PRV</given-names>
</name>
</person-group>. <article-title>Characterization of complex networks: a survey of measurements</article-title>. <source>Adv Phys</source> (<year>2007</year>) <volume>56</volume>:<fpage>167</fpage>&#x2013;<lpage>242</lpage>. <pub-id pub-id-type="doi">10.1080/00018730601170527</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>James</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Menzies</surname>
<given-names>M</given-names>
</name>
</person-group>. <article-title>Equivalence relations and <italic>l</italic>
<sup>p</sup> distances between time series</article-title>. <source>Physica D Nonlinear Phenomena</source> (<year>2020</year>) <volume>448</volume>:<fpage>133693</fpage>. <pub-id pub-id-type="doi">10.1016/j.physd.2023.133693</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>