<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2024.1397915</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Multi-scale dynamics by adjusting the leaking rate to enhance the performance of deep echo state networks</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Inoue</surname> <given-names>Shuichi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2776374/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Nobukawa</surname> <given-names>Sou</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/540470/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Nishimura</surname> <given-names>Haruhiko</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/943215/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Watanabe</surname> <given-names>Eiji</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<xref ref-type="aff" rid="aff8"><sup>8</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/506564/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Isokawa</surname> <given-names>Teijiro</given-names></name>
<xref ref-type="aff" rid="aff9"><sup>9</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1851294/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Graduate School of Information and Computer Science, Chiba Institute of Technology</institution>, <addr-line>Narashino</addr-line>, <country>Japan</country></aff>
<aff id="aff2"><sup>2</sup><institution>LY Corporation</institution>, <addr-line>Chiyoda-ku</addr-line>, <country>Japan</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Computer Science, Chiba Institute of Technology</institution>, <addr-line>Narashino</addr-line>, <country>Japan</country></aff>
<aff id="aff4"><sup>4</sup><institution>Research Center for Mathematical Engineering, Chiba Institute of Technology</institution>, <addr-line>Narashino</addr-line>, <country>Japan</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Preventive Intervention for Psychiatric Disorders, National Institute of Mental Health, National Center of Neurology and Psychiatry</institution>, <addr-line>Kodaira</addr-line>, <country>Japan</country></aff>
<aff id="aff6"><sup>6</sup><institution>Faculty of Informatics, Yamato University</institution>, <addr-line>Osaka</addr-line>, <country>Japan</country></aff>
<aff id="aff7"><sup>7</sup><institution>Laboratory of Neurophysiology, National Institute for Basic Biology</institution>, <addr-line>Okazaki</addr-line>, <country>Japan</country></aff>
<aff id="aff8"><sup>8</sup><institution>Department of Basic Biology, The Graduate University for Advanced Studies</institution>, <addr-line>Hayama</addr-line>, <country>Japan</country></aff>
<aff id="aff9"><sup>9</sup><institution>Graduate School of Engineering, University of Hyogo</institution>, <addr-line>Himeji</addr-line>, <country>Japan</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Chao Luo, Shandong Normal University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Raymond Lee, Beijing Normal University-Hong Kong Baptist University United International College, China</p>
<p>Guilherme De Alencar Barreto, Federal University of Ceara, Brazil</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Sou Nobukawa <email>nobukawa&#x00040;cs.it-chiba.ac.jp</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>07</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>7</volume>
<elocation-id>1397915</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>06</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Inoue, Nobukawa, Nishimura, Watanabe and Isokawa.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Inoue, Nobukawa, Nishimura, Watanabe and Isokawa</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>The deep echo state network (Deep-ESN) architecture, which comprises a multi-layered reservoir layer, exhibits superior performance compared to conventional echo state networks (ESNs) owing to the divergent layer-specific time-scale responses in the Deep-ESN. Although researchers have attempted to use experimental trial-and-error grid searches and Bayesian optimization methods to adjust the hyperparameters, suitable guidelines for setting hyperparameters to adjust the time scale of the dynamics in each layer from the perspective of dynamical characteristics have not been established. In this context, we hypothesized that evaluating the dependence of the multi-time-scale dynamical response on the leaking rate as a typical hyperparameter of the time scale in each neuron would help to achieve a guideline for optimizing the hyperparameters of the Deep-ESN.</p></sec>
<sec>
<title>Method</title>
<p>First, we set several leaking rates for each layer of the Deep-ESN and performed multi-scale entropy (MSCE) analysis to analyze the impact of the leaking rate on the dynamics in each layer. Second, we performed layer-by-layer cross-correlation analysis between adjacent layers to elucidate the structural mechanisms to enhance the performance.</p></sec>
<sec>
<title>Results</title>
<p>As a result, an optimum task-specific leaking rate value for producing layer-specific multi-time-scale responses and a queue structure with layer-to-layer signal transmission delays for retaining past applied input enhance the Deep-ESN prediction performance.</p></sec>
<sec>
<title>Discussion</title>
<p>These findings can help to establish ideal design guidelines for setting the hyperparameters of Deep-ESNs.</p></sec></abstract>
<kwd-group>
<kwd>multi-scale dynamics</kwd>
<kwd>machine learning</kwd>
<kwd>reservoir computing</kwd>
<kwd>echo state network</kwd>
<kwd>deep echo state network</kwd>
</kwd-group>
<contract-num rid="cn001">JP20H05921</contract-num>
<contract-num rid="cn001">JP22K12183</contract-num>
<contract-sponsor id="cn001">Japan Society for the Promotion of Science<named-content content-type="fundref-id">10.13039/501100001691</named-content></contract-sponsor>
<counts>
<fig-count count="10"/>
<table-count count="0"/>
<equation-count count="12"/>
<ref-count count="37"/>
<page-count count="13"/>
<word-count count="7031"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Reservoir computing, which is a branch of recurrent neural networks (RNNs), has garnered significant interest in terms of machine-learning applications (Luko&#x00161;evi&#x0010D;ius and Jaeger, <xref ref-type="bibr" rid="B21">2009</xref>; Tanaka et al., <xref ref-type="bibr" rid="B31">2019</xref>; Gallicchio and Micheli, <xref ref-type="bibr" rid="B10">2021</xref>). A neural network for reservoir computing consists of three layers: an input layer, a reservoir layer, and an output layer (Jaeger, <xref ref-type="bibr" rid="B16">2001</xref>; Luko&#x00161;evi&#x0010D;ius and Jaeger, <xref ref-type="bibr" rid="B21">2009</xref>). In reservoir computing, the input time-series data are transformed into spatio-temporal patterns in the reservoir layer. The responses of individual neurons act as a kernel for these transformed patterns, representing the desired output signal, which is consequently used for time-series prediction and classification (Jaeger, <xref ref-type="bibr" rid="B16">2001</xref>; Gallicchio and Micheli, <xref ref-type="bibr" rid="B10">2021</xref>).</p>
<p>The echo state network (ESN), which is a representative model in reservoir computing, operates based on the response of the firing rate (Jaeger, <xref ref-type="bibr" rid="B17">2007</xref>). In an ESN, as depicted in <xref ref-type="fig" rid="F1">Figure 1</xref>, the synaptic connections of the reservoir weights are fixed, and only the synaptic connections of the output weight matrix are adjusted in the learning process (Tanaka et al., <xref ref-type="bibr" rid="B31">2019</xref>). This architecture differs from that of other RNNs, wherein all synaptic connections within the network undergo updates during learning (Williams and Zipser, <xref ref-type="bibr" rid="B36">1989</xref>). Thus, ESNs are more learning efficient (Werbos, <xref ref-type="bibr" rid="B35">1990</xref>) than the widely utilized long-short-term memory model, despite exhibiting lower accuracy (Salehinejad et al., <xref ref-type="bibr" rid="B27">2017</xref>; Gallicchio et al., <xref ref-type="bibr" rid="B12">2018</xref>). Such efficient learning architectures may offer the potential for applications in areas that are characterized by resource-limited hardware, such as edge devices (Tanaka et al., <xref ref-type="bibr" rid="B31">2019</xref>; Sakemi et al., <xref ref-type="bibr" rid="B26">2024</xref>).</p>


<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Architecture of ESN.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1397915-g0001.tif"/>
</fig>


<p>The deep echo state network (Deep-ESN) model, which possesses the multi-layered reservoir structure illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>, has also been proposed. This model performs better than the conventional ESN, which consists of a single-layered reservoir (Deng et al., <xref ref-type="bibr" rid="B8">2012</xref>; Gallicchio et al., <xref ref-type="bibr" rid="B11">2017</xref>; Gallicchio and Micheli, <xref ref-type="bibr" rid="B10">2021</xref>). The divergent responses in each layer of the Deep-ESN, which exhibits multiple time-scale dynamics, enhance the memory capacity and feature representation compared to its single-layered counterpart (Malik et al., <xref ref-type="bibr" rid="B23">2016</xref>; Gallicchio et al., <xref ref-type="bibr" rid="B11">2017</xref>; Tchakoucht and Ezziyyani, <xref ref-type="bibr" rid="B32">2018</xref>; Gallicchio and Micheli, <xref ref-type="bibr" rid="B9">2019</xref>; Long et al., <xref ref-type="bibr" rid="B20">2019</xref>; Kanda and Nobukawa, <xref ref-type="bibr" rid="B19">2022</xref>). These advantages of the Deep-ESN may enable applications in tasks involving nonlinear dynamic signals that exhibit multi-time-scale behaviors in various types of systems, such as biological systems, power systems, and financial markets (Venkatasubramanian et al., <xref ref-type="bibr" rid="B33">1995</xref>; Costa et al., <xref ref-type="bibr" rid="B6">2002</xref>; Bhandari, <xref ref-type="bibr" rid="B4">2017</xref>; Chen and Shang, <xref ref-type="bibr" rid="B5">2020</xref>; Yan and He, <xref ref-type="bibr" rid="B37">2021</xref>).</p>






<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Architecture of Deep-ESN.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1397915-g0002.tif"/>
</fig>


<p>The adjustment of numerous hyperparameters in the Deep-ESN is often based on experimental measurements, trial-and-error grid searches, and Bayesian optimization methods (Adeleke, <xref ref-type="bibr" rid="B1">2019</xref>; Luko&#x00161;evi&#x0010D;ius and Uselis, <xref ref-type="bibr" rid="B22">2019</xref>; Bai et al., <xref ref-type="bibr" rid="B3">2023</xref>; Viehweg et al., <xref ref-type="bibr" rid="B34">2023</xref>). In terms of Bayesian optimization in Deep-ESNs, emphasis is mainly placed on performance, and the analysis of the reservoir dynamics and mechanisms of functionality enhancement is often overlooked; consequently, no specific design guidelines on the dynamics have been presented (Bai et al., <xref ref-type="bibr" rid="B3">2023</xref>). In terms of hyperparameters that adjust the time scale of each layer, several studies have applied scaling methods to the input weight matrix <italic>W</italic><sup>(<italic>l</italic>)</sup>, as illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref> (Kanda and Nobukawa, <xref ref-type="bibr" rid="B19">2022</xref>), to achieve the establishment of guidelines. By integrating this method, the signal strength between layers decreases as the depth increases, which results in a layer-specific time-scale response for each layer (Kanda and Nobukawa, <xref ref-type="bibr" rid="B19">2022</xref>). This characteristic has been revealed by analyzing the layer dynamics using the multi-scale entropy (MSCE) method (Humeau-Heurtier, <xref ref-type="bibr" rid="B14">2015</xref>; Kanda and Nobukawa, <xref ref-type="bibr" rid="B19">2022</xref>). However, this approach, in which the input weights are scaled, has a major drawback. The signal strength in the reservoir layer diminishes quickly owing to the small scaling rate between layers. In addition, the leaking rate is a hyperparameter that influences the temporal history effect of the dynamics in <bold>x</bold><sup>(<italic>l</italic>)</sup>(<italic>t</italic>) of <xref ref-type="fig" rid="F2">Figure 2</xref> (Jaeger et al., <xref ref-type="bibr" rid="B18">2007</xref>; Schrauwen et al., <xref ref-type="bibr" rid="B28">2007</xref>). Essentially, the leaking rate adjusts the decay factor of the dynamics in each neuron (Jaeger et al., <xref ref-type="bibr" rid="B18">2007</xref>; Schrauwen et al., <xref ref-type="bibr" rid="B28">2007</xref>). Therefore, the method for adjusting the leaking rate is another suitable candidate for achieving the layer-specific dynamical response in the Deep-ESN.</p>
<p>In this context, we hypothesized that evaluating the dependence of the dynamical response in the multi-layered reservoir in terms of adjusting the leaking rate would provide insights into achieving a guideline for optimizing the hyperparameters of the Deep-ESN. For the preliminary investigation, we set the same leaking rate for each layer of the Deep-ESN and performed an MSCE analysis to investigate the impact of the leaking rate on the dynamics in each layer. The results confirmed that each layer of the Deep-ESN generates dynamics at different time scales, which induces a queue-like property whereby the delay response is preserved by the hierarchical structure (Inoue et al., <xref ref-type="bibr" rid="B15">2023</xref>). However, the performance tendencies for more diverse time-series prediction tasks remain unclear; an evaluation in the case of setting heterogeneous leaking rates in the multi-layered reservoir has not been conducted. Therefore, in this study, based on the preliminary outcomes of a previous study (Inoue et al., <xref ref-type="bibr" rid="B15">2023</xref>), we further revealed these points. Specifically, Deep-ESNs with homogenous and heterogeneous leaking rates for each layer were used to perform and evaluate a time-series prediction task using three time-series signals: the Lorenz, R&#x000F6;ssler, and Mackey&#x02013;Glass models. Furthermore, we performed layer-by-layer MSCE and cross-correlation analyses between adjacent layers to elucidate the mechanisms behind the functional enhancement achieved through the leaking rates.</p></sec>
<sec sec-type="materials and methods" id="s2">
<title>2 Materials and methods</title>
<sec>
<title>2.1 ESN</title>
<p><xref ref-type="fig" rid="F1">Figure 1 </xref>shows the architecture of the ESN. The input signal is defined as <inline-formula><mml:math id="M1"><mml:mstyle mathvariant="bold"><mml:mtext>u</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> with <italic>N</italic><sub><italic>u</italic></sub>-dimensional inputs. The reservoir state is defined by <inline-formula><mml:math id="M2"><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>, where <italic>N</italic><sub><italic>x</italic></sub> is the number of neurons in the reservoir layer. The reservoir state <bold>x</bold>(<italic>t</italic>) is defined by</p>
<disp-formula id="E1"><mml:math id="M3"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant='bold'><mml:mtext>x</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>a</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mstyle mathvariant='bold'><mml:mtext>x</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>a</mml:mi><mml:mo class="qopname">tanh</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant='bold'><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:msub><mml:mstyle mathvariant="bold"><mml:mtext>u</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant='bold'><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover><mml:mstyle mathvariant='bold'><mml:mtext>x</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>a</italic>&#x02208;[0, 1] is the leaking rate and <inline-formula><mml:math id="M4"><mml:msub><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is the input weight matrix. Each component of <bold>W</bold><sub>in</sub> is represented by a uniform random value, the range of which is [&#x02212;<italic>s</italic><sub>in</sub>, <italic>s</italic><sub>in</sub>]. <inline-formula><mml:math id="M5"><mml:mover accent="true"><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">W</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is the recurrent synaptic weight matrix, which is a random matrix with uniform random numbers, and its spectral radius is set to &#x003C1;. The output of the ESN at time <italic>t</italic> is determined using</p>
<disp-formula id="E2"><mml:math id="M6"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant='bold'><mml:mtext>y</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant='bold'><mml:mtext>W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub><mml:mstyle mathvariant='bold'><mml:mtext>x</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M7"><mml:msub><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is defined as the output weight matrix and <inline-formula><mml:math id="M8"><mml:mstyle class="text"><mml:mtext mathvariant="bold">y</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is the <italic>N</italic><sub><italic>y</italic></sub>-dimensional output. The initial value of <bold>W</bold><sub>out</sub> is a random matrix of uniform random numbers.</p></sec>
<sec>
<title>2.2 Deep-ESN</title>
<p>The multi-layered Deep-ESN is constructed based on the single-layered ESN; <xref ref-type="fig" rid="F2">Figure 2</xref> exhibits a diagram of the Deep-ESN (Deng et al., <xref ref-type="bibr" rid="B8">2012</xref>; Gallicchio et al., <xref ref-type="bibr" rid="B11">2017</xref>; Gallicchio and Micheli, <xref ref-type="bibr" rid="B10">2021</xref>). The only difference from the ESN is that the reservoir layer is multi-layered. The reservoir state vector of the Deep-ESN is defined by</p>
<disp-formula id="E3"><mml:math id="M9"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>a</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:msup><mml:mi>a</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mi>tanh</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>W</mml:mi></mml:mstyle><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>i</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>+</mml:mo><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mover accent='true'><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>W</mml:mi></mml:mstyle><mml:mo stretchy='true'>&#x0005E;</mml:mo></mml:mover><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where variables consisting of <italic>l</italic> refer to <italic>l</italic>-layer parameters. <inline-formula><mml:math id="M10"><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>&#x003B8;</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is the bias in the reservoir coupling weight matrix. <inline-formula><mml:math id="M11"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is the input weight matrix for each layer, and <inline-formula><mml:math id="M12"><mml:msup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">W</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is the recurrent weight matrix for each layer. The dynamics of each reservoir layer can be defined as <inline-formula><mml:math id="M13"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula> by averaging the reservoir state vector <bold>x</bold><sup>(<italic>l</italic>)</sup>(<italic>t</italic>) over all neurons. In addition, <bold>i</bold><sup>(<italic>l</italic>)</sup>(<italic>t</italic>) represents the input to the <italic>l</italic>-th layer in the Deep-ESN, which is expressed as</p>
<disp-formula id="E4"><mml:math id="M14"><mml:mrow><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>i</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>u</mml:mi></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>if&#x000A0;</mml:mtext><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>if&#x000A0;</mml:mtext><mml:mi>l</mml:mi><mml:mo>&#x0003E;</mml:mo><mml:mn>1.</mml:mn></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>The output of the Deep-ESN at time <italic>t</italic> is determined by</p>
<disp-formula id="E5"><mml:math id="M15"><mml:mrow><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>W</mml:mi></mml:mstyle><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mo stretchy='false'>[</mml:mo><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>2</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mn>...</mml:mn><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:msup><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mi>L</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>]</mml:mo></mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:msup><mml:mo>+</mml:mo><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>&#x003B8;</mml:mi></mml:mstyle><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where <italic>N</italic><sub><italic>L</italic></sub> is defined as the total number of reservoir layers and <inline-formula><mml:math id="M16"><mml:mstyle class="text"><mml:mtext mathvariant="bold">y</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is the output in <italic>N</italic><sub><italic>y</italic></sub> dimensions. <inline-formula><mml:math id="M17"><mml:msub><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">W</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is the output weight matrix. The bias in the output layer is set to <inline-formula><mml:math id="M18"><mml:mrow><mml:msub><mml:mi>&#x003B8;</mml:mi><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo stretchy='false'>[</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>...</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>]</mml:mo></mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:msup></mml:mrow></mml:math></inline-formula>.</p>
<p>In this study, the total number of layers was set to <italic>N</italic><sub><italic>L</italic></sub> &#x0003D; 10, the number of neurons in each layer was set to <italic>N</italic><sub><italic>x</italic></sub> &#x0003D; 100, the scaling parameter of the input weight matrix <bold>W</bold><sub>in</sub> was set to <italic>s</italic><sub>in</sub> &#x0003D; 1, the spectral radius was set to &#x003C1; &#x0003D; 1.0, 0.9, 0.8, and ridge regression was used as the learning algorithm. For the settings of the leaking rate <italic>a</italic><sup>(<italic>l</italic>)</sup> in this experiment, a homogeneous model with the same leaking rate in all layers and a heterogeneous model with different leaking rates in each layer were used. The leaking rate <italic>a</italic><sup>(<italic>l</italic>)</sup> of the homogeneous model was set to be <italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 1.0, 0.9, ..., 0.1 in all layers. In the heterogeneous model, the leaking rate <italic>a</italic><sup>(<italic>l</italic>)</sup> was set to decrease incrementally by 0.1 across each layer, starting from 1.0 and decreasing to 0.1. For simplicity, the same values were set for the hyperparameters <italic>N</italic><sub><italic>L</italic></sub>, <italic>N</italic><sub><italic>x</italic></sub>, <italic>s</italic><sub>in</sub>, &#x003C1;, and <italic>a</italic><sup>(<italic>l</italic>)</sup> for each layer in the reservoir. In addition, <bold>W</bold><sub>in</sub>, <inline-formula><mml:math id="M19"><mml:msup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">W</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:math></inline-formula>, <bold>W</bold><sub>out</sub>, and <bold>x</bold>(0) were initialized with different random seeds for each trial. The seed values were changed during the execution of the time-series prediction task, and 100 trials were performed.</p></sec>
<sec>
<title>2.3 Time-series prediction task</title>
<p>In terms of the impact of the leaking rate on the performance, the Mackey&#x02013;Glass, Lorenz, and R&#x000F6;ssler equations were prepared as time-series signals with different dynamical characteristics. In this study, homogeneous and heterogeneous models were used in the time-series prediction task, which was evaluated by predicting the five-step-ahead value from the current input. The input signal was continuously applied to the models. Each task involved 100 trials with different initial values. The normalized root mean square error (NRMSE) was used to evaluate the prediction accuracy for each task in the homogeneous and heterogeneous models.</p>
<sec>
<title>2.3.1 Mackey&#x02013;Glass equation</title>
<p>For the time-series prediction, time series were generated from the Mackey&#x02013;Glass equation (Glass and Mackey, <xref ref-type="bibr" rid="B13">2010</xref>):</p>
<disp-formula id="E6"><mml:math id="M20"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">mg</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>2</mml:mn><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">mg</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">mg</mml:mtext></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mo>-</mml:mo><mml:mn>0</mml:mn><mml:mo>.</mml:mo><mml:mn>1</mml:mn><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">mg</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003C4; is a constant representing the delay. In this study, under the condition of &#x003C4; &#x0003D; 32, 64, and 128, the solution was obtained using the fourth-order Runge&#x02013;Kutta method, and the trajectories were sampled in a time window of &#x00394;<italic>t</italic> &#x0003D; 10.</p></sec>
<sec>
<title>2.3.2 Lorenz equation</title>
<p>The Lorenz equation is represented by a system of nonlinear differential equations of the form (Manneville and Pomeau, <xref ref-type="bibr" rid="B24">1979</xref>)</p>
<disp-formula id="E7"><mml:math id="M21"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:msub><mml:mi>x</mml:mi><mml:mtext>l</mml:mtext></mml:msub></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mi>&#x003C3;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mtext>l</mml:mtext></mml:msub><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mtext>l</mml:mtext></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:msub><mml:mi>y</mml:mi><mml:mtext>l</mml:mtext></mml:msub></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mtext>l</mml:mtext></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mtext>l</mml:mtext></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mtext>l</mml:mtext></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:msub><mml:mi>z</mml:mi><mml:mtext>l</mml:mtext></mml:msub></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mtext>l</mml:mtext></mml:msub><mml:msub><mml:mi>y</mml:mi><mml:mtext>l</mml:mtext></mml:msub><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B2;</mml:mi><mml:msub><mml:mi>z</mml:mi><mml:mtext>l</mml:mtext></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The parameters for the Lorenz equation were &#x003C3; &#x0003D; 10, <italic>r</italic> &#x0003D; 28, and <italic>b</italic> &#x0003D; 8/3. These values are known to exhibit chaotic behavior. In this study, the solution was obtained using the fourth-order Runge&#x02013;Kutta method, and the trajectories were sampled in the time window &#x00394;<italic>t</italic> &#x0003D; 0.02.</p></sec>
<sec>
<title>2.3.3 R&#x000F6;ssler equation</title>
<p>The R&#x000F6;ssler system, which is a nonlinear dynamical system (R&#x000F6;ssler, <xref ref-type="bibr" rid="B25">1983</xref>), was adopted to generate chaotic time-series data. The system is defined by the following set of three nonlinear ordinary differential equations:</p>
<disp-formula id="E8"><mml:math id="M22"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">r</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">r</mml:mtext></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">r</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E9"><mml:math id="M23"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">r</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">r</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>a</mml:mi><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">r</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">r</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mi>b</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">r</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">r</mml:mtext></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mi>c</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In these equations, <italic>x</italic><sub>r</sub>, <italic>y</italic><sub>r</sub>, and <italic>z</italic><sub>r</sub> represent the system states. The parameters <italic>a</italic>, <italic>b</italic>, and <italic>c</italic> directly affect the system&#x00027;s behavior. For our experiments, the parameters were set to <italic>a</italic> &#x0003D; 0.2, <italic>b</italic> &#x0003D; 0.2, and <italic>c</italic> &#x0003D; 5.7. In this study, the solution was obtained using the fourth-order Runge&#x02013;Kutta method, and the trajectories were sampled in the time window &#x00394;<italic>t</italic> &#x0003D; 0.02.</p></sec></sec>
<sec>
<title>2.4 MSCE analysis</title>
<p>MSCE analysis is a method for performing coarse-graining of the time series of interest and quantitatively evaluating the complexity across multiple time scales (Humeau-Heurtier, <xref ref-type="bibr" rid="B14">2015</xref>). As an analytical procedure, the first step is to coarse-grain the dynamics of each reservoir layer <inline-formula><mml:math id="M24"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula> with a time-scale factor &#x003C4;<sub><italic>s</italic></sub>, using</p>
<disp-formula id="E10"><mml:math id="M25"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover><mml:mo>,</mml:mo><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x02264;</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x02264;</mml:mo><mml:mfrac><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In the case of &#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1, the original time series is coarse-grained to longer-scale dynamics as &#x003C4;<sub><italic>s</italic></sub> increases. At each time scale &#x003C4;<sub><italic>s</italic></sub> and layer <italic>l</italic>, the complexity of the coarse-grained time series is then quantified by the sample entropy (SampEn). SampEn is given by</p>
<disp-formula id="E11"><mml:math id="M26"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">SampEn</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>N</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mo class="qopname">log</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>U</italic><sub><italic>m</italic></sub>(<italic>r</italic>) represents the probability of being <inline-formula><mml:math id="M27"><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>|</mml:mo><mml:mo>&#x0003C;</mml:mo><mml:mi>r</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02260;</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M28"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> represents the <italic>m</italic>-dimensional vector <inline-formula><mml:math id="M29"><mml:msubsup><mml:mrow><mml:mstyle class="text"><mml:mtext mathvariant="bold">z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>. Thus, the complexity of the dynamics of each layer can be analyzed and evaluated from different time-scale perspectives.</p></sec>
<sec>
<title>2.5 NRMSE</title>
<p>The NRMSE is a statistical measure that is used to assess the accuracy of a model&#x00027;s predictions and is defined as follows:</p>
<disp-formula id="E12"><mml:math id="M30"><mml:mrow><mml:mtext>NRMSE</mml:mtext><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mfrac><mml:mrow><mml:mstyle displaystyle='true'><mml:msubsup><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>T</mml:mi></mml:msubsup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy='false'>(</mml:mo></mml:mrow></mml:mstyle><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>d</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:msup><mml:mi>&#x003C3;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>d</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac></mml:mrow></mml:msqrt><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
<p>In this study, the task inputs and outputs were one-dimensional. Therefore, <italic>y</italic>(<italic>t</italic>) is the output of the ESN in the case with <italic>N</italic><sub><italic>y</italic></sub> &#x0003D; 1 at time <italic>t</italic>, <italic>y</italic><sub><italic>d</italic></sub>(<italic>t</italic>) is the teacher signal at time <italic>t</italic>, &#x003C3;<sup>2</sup>(<italic>y</italic>(<italic>t</italic>)) is the variance of the teacher signal, and <italic>T</italic> is the evaluation period (number of data points). The NRMSE was evaluated among 100 trials with different initial conditions.</p></sec>
<sec>
<title>2.6 Cross-correlation analysis</title>
<p>Cross-correlation analysis, which evaluates the synchronization with a delay between two time-series signals, is extensively used for the signal transmission of neural activity in hierarchical brain networks (Adhikari et al., <xref ref-type="bibr" rid="B2">2010</xref>; Dean and Dunsmuir, <xref ref-type="bibr" rid="B7">2016</xref>). Thus, we adopted the cross-correlation analysis for the signal transmission in the Deep-ESN. We used the cross-correlation Corr(<italic>k</italic>) between the dynamics of the reservoir state at the <italic>l</italic>-th and <italic>l</italic>&#x0002B;1-th layers, i.e., the time-series <inline-formula><mml:math id="M31"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula> and <inline-formula><mml:math id="M32"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula>, where <italic>k</italic> is the delay time and each time series is <italic>z</italic>-score transformed.</p></sec>
<sec>
<title>2.7 Experimental procedures</title>
<p>This study implemented the following procedures to explore parameters that demonstrate dynamics that can achieve high performance in Deep-ESNs: We initially employed MSCE analysis to identify the time-scale dependency of the dynamical responses in terms of complexity among the layers. This analysis aims to guide the optimization of Deep-ESNs for enhanced handling of time-series prediction tasks. After completing the MSCE analysis, we proceeded with the synchronization analysis to detect delays between adjacent layers. This sequential approach facilitates a comprehensive understanding of both the dynamics of each layer and the interactions between layers. Through these analyses, our focus shifted to establishing guidelines for setting hyperparameters, particularly in cases where previous studies on Deep-ESNs did not yield significant findings on the dynamical characteristics. These guidelines are designed to refine the time scale of the dynamics in Deep-ESNs based on the insights gained from our MSCE and cross-correlation analyses.</p></sec></sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec>
<title>3.1 Time-series prediction task</title>
<p>We evaluated the dependence of the performance of Deep-ESNs on the leaking rates <italic>a</italic><sup>(<italic>l</italic>)</sup> in the time-series prediction tasks in the nonlinear dynamical signals of the Mackey&#x02013;Glass, Lorenz, and R&#x000F6;ssler equations. <xref ref-type="fig" rid="F3">Figures 3</xref>&#x02013;<xref ref-type="fig" rid="F5">5</xref> convey the results of the evaluation of the homogeneous and heterogeneous models in each time-series prediction task. In the homogeneous model, the leaking rate was set to <italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 1.0, 0.9, ..., 0.1 for all layers. In the heterogeneous model, the leaking rate decreased by 0.1 in each layer, from 1.0 to 0.1. The dependences of the NRMSE on the leaking rate in the homogeneous model for the spectral radii &#x003C1; &#x0003D; 0.8, 0.9, and 1.0 are displayed in the left panels of <xref ref-type="fig" rid="F3">Figures 3</xref>&#x02013;<xref ref-type="fig" rid="F5">5</xref>. The results demonstrate that the profiles of the NRMSEs in all tasks and the spectral radii exhibited a <italic>U</italic>-shape in response to the leaking rate, indicating the presence of an optimal leaking rate for the prediction tasks. Furthermore, the right panels of <xref ref-type="fig" rid="F3">Figures 3</xref>&#x02013;<xref ref-type="fig" rid="F5">5</xref> present a comparison of the NRMSEs of the most superior cases in the homogeneous model, which were obtained from the profile of the dependence on the leaking rate in the left panels, with the heterogeneous model leaking rate. The results reveal that the NRMSE of the heterogeneous model was significantly low (based on the paired-<italic>t</italic> test using a Bonferroni false discovery rate with <italic>q</italic> &#x0003C; 0.05) only for the Mackey&#x02013;Glass time series (&#x003C4; &#x0003D; 64) across all spectral radii. This tendency implied that the heterogeneous models could precisely respond to strong multi-temporal-scale dynamics, similar to the Mackey&#x02013;Glass time series (&#x003C4; &#x0003D; 64) that exhibited wide multi-temporal-scale components in this dynamics (see Section 1 of <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>).</p>














<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Prediction performance in the Mackey&#x02013;Glass time series. (Left panel) Dependence of the performance of the Deep-ESN on leaking rates (<italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 1.0, 0.9, ..., 0.1 for all layers) in the homogeneous (hm) model with spectral radii &#x003C1; &#x0003D; 1.0 <bold>(A)</bold>, 0.9 <bold>(B)</bold>, and 0.8 <bold>(C)</bold>. The profiles of the NRMSEs in all spectral radii exhibited a <italic>U</italic>-shape against the leaking rate, indicating the presence of an optimal leaking rate for the prediction tasks. (Right panel) NRMSEs for the most superior cases in the homogeneous model (corresponding to the cases with minimum NRMSEs presented in the left panel) and for the cases of the heterogeneous (het) model in which the leaking rate was set to decrease by 0.1 in each layer, from 1.0 to 0.1. The NRMSE of the heterogeneous model was low [based on the paired-<italic>t</italic> test using a Bonferroni false discovery rate with <italic>q</italic> &#x0003C; 0.05 (<italic>p</italic> &#x0003C; 0.05/9)] only for the Mackey&#x02013;Glass time series (&#x003C4; &#x0003D; 64) throughout all spectral radii.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1397915-g0003.tif"/>
</fig>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Prediction performance in the Lorenz time series. (Left panel) Dependence of the performance of the Deep-ESN on leaking rates [<italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 1.0, 0.9, ..., 0.1 for all layers] in the homogeneous (hm) model with spectral radii &#x003C1; &#x0003D; 1.0 <bold>(A)</bold>, 0.9 <bold>(B)</bold>, and 0.8 <bold>(C)</bold>. The profiles of the NRMSEs in all spectral radii exhibited a <italic>U</italic>-shape in response to the leaking rate, indicating the presence of an optimal leaking rate for the prediction tasks. (Right panel) NRMSEs for the most superior cases in the homogeneous model (corresponding to the cases with the minimum NRMSEs highlighted in the left panel) and for the cases of the heterogeneous (het) model in which the leaking rate was set to decrease by 0.1 in each layer, from 1.0 to 0.1. The NRMSE of the homogeneous model was low [based on the paired-<italic>t</italic> test using a Bonferroni false discovery rate with <italic>q</italic> &#x0003C; 0.05 (<italic>p</italic> &#x0003C; 0.05/9)] for the Lorenz time series [<italic>x</italic><sub>l</sub>(<italic>t</italic>) and <italic>y</italic><sub>l</sub>(<italic>t</italic>)] throughout all spectral radii.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1397915-g0004.tif"/>
</fig>

<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Prediction performance in the R&#x000F6;ssler time series. (Left panel) Dependence of the performance of the Deep-ESN on leaking rates [<italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 1.0, 0.9, ..., 0.1 for all layers] in the homogeneous (hm) model with spectral radii &#x003C1; &#x0003D; 1.0 <bold>(A)</bold>, 0.9 <bold>(B)</bold>, and 0.8 <bold>(C)</bold>. The profiles of the NRMSEs in all spectral radii exhibited a <italic>U</italic>-shape against the leaking rate, indicating the presence of an optimal leaking rate for the prediction tasks. (Right panel) NRMSEs for the most superior cases in the homogeneous model (corresponding to the cases with the minimum NRMSEs shown in the left panel) and for the cases of the heterogeneous (het) model in which the leaking rate was set to decrease by 0.1 in each layer, from 1.0 to 0.1. The NRMSE of the homogeneous model was low [based on the paired-<italic>t</italic> test using a Bonferroni false discovery rate with <italic>q</italic> &#x0003C; 0.05 (<italic>p</italic> &#x0003C; 0.05/9)] for the R&#x000F6;ssler time series [<italic>x</italic><sub>l</sub>(<italic>t</italic>), <italic>y</italic><sub>l</sub>(<italic>t</italic>) and <italic>z</italic><sub>l</sub>(<italic>t</italic>)] throughout all spectral radii.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1397915-g0005.tif"/>
</fig>
</sec>


<sec>
<title>3.2 MSCE analysis</title>
<p>MSCE analysis was used to investigate the behavior of the dynamics of each layer of the reservoir. <xref ref-type="fig" rid="F6">Figures 6</xref>&#x02013;<xref ref-type="fig" rid="F8">8</xref> exhibit the results of the MSCE analysis for the Mackey&#x02013;Glass, Lorenz, and R&#x000F6;ssler tasks, respectively. The temporal scales &#x003C4;<sub><italic>s</italic></sub> were set to 1, 10, and 20. In the case of the homogeneous model, a trend toward reduced complexity on the fast time scale (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1) was observed as the leaking rate was reduced. The complexity with slow time scales (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 10, 20) may vary across layers or be almost constant, depending on the time scale of the task and leaking rate. In the heterogeneous model case, the complexity on the fast time scale (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1) tended to decrease with the layer depth. The complexity on the slow time scales (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 10, 20) tended to increase layer by layer.</p>








<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>MSCE analysis of reservoir dynamics (temporal scale &#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1, 10, 20) with spectral radius &#x003C1; &#x0003D; 1.0 in the case of the Mackey&#x02013;Glass task. The first, second, and third columns list the cases with Mackey&#x02013;Glass delay time constants &#x003C4; &#x0003D; 32, 64, and 128, respectively. The first, second, and third columns correspond to the maximum value of the leaking rate [<italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 1.0], the leaking rate when the NRMSE was the best, and the minimum leaking rate [<italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 0.1], respectively. In the case of the homogeneous model, a trend toward reduced complexity on the fast time scale (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1) was observed as the leaking rate was reduced. The complexity on the slow time scales (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 10, 20) tended to vary across layers or to be almost constant depending on the time scale of the task and leaking rate. In the heterogeneous model case, the complexity on the fast time scale (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1) tended to decrease according to the layer depth. The complexity on the slow time scale (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 10, 20) tended to increase layer by layer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1397915-g0006.tif"/>
</fig>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>MSCE analysis of reservoir dynamics (temporal scale &#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1, 10, 20) with spectral radius &#x003C1; &#x0003D; 1.0 in the case of the Lorenz task. The first, second, and third columns display the cases with Lorenz tasks <italic>x</italic><sub>l</sub>(<italic>t</italic>), <italic>y</italic><sub>l</sub>(<italic>t</italic>), and <italic>z</italic><sub>l</sub>(<italic>t</italic>), respectively. The first, second, and third columns correspond to the maximum value of the leaking rate [<italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 1.0], the leaking rate when the NRMSE was the best, and the minimum value of the leaking rate [<italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 0.1], respectively. In the case of the homogeneous model, a trend toward reduced complexity on the fast time scale (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1) was observed as the leaking rate was reduced. The complexity on the slow time scales (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 10, 20) tended to vary across layers or to be almost constant, depending on <italic>x</italic><sub><italic>l</italic></sub>/<italic>y</italic><sub><italic>l</italic></sub>/<italic>z</italic><sub><italic>l</italic></sub> and the leaking rate. In the heterogeneous model case, the complexity on the fast time scale (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1) tended to decrease according to the layer depth. The complexity on the slow time scales (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 10, 20) tended to increase layer by layer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1397915-g0007.tif"/>
</fig>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>MSCE analysis of reservoir dynamics (temporal scale &#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1, 10, 20) with spectral radius &#x003C1; &#x0003D; 1.0 in the case of the R&#x000F6;ssler task. The first, second, and third columns convey the cases with R&#x000F6;ssler tasks <italic>x</italic><sub>r</sub>(<italic>t</italic>), <italic>y</italic><sub>r</sub>(<italic>t</italic>), and <italic>z</italic><sub>r</sub>(<italic>t</italic>), respectively. The first, second, and third columns correspond to the maximum value of the leaking rate [<italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 1.0], the leaking rate when the NRMSE was the best, and the minimum value of the leaking rate [<italic>a</italic><sup>(<italic>l</italic>)</sup> &#x0003D; 0.1], respectively. In the homogeneous case with task <italic>x</italic><sub>r</sub>(<italic>t</italic>), <italic>y</italic><sub>r</sub>(<italic>t</italic>), in the case with a high leaking rate, the complexity of the fast time scale (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1) increased with the layer depths. The complexity on the slow time scales (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 10, 20) tended to vary across layers or to be almost constant, depending on <italic>x</italic><sub><italic>r</italic></sub>/<italic>y</italic><sub><italic>r</italic></sub>/<italic>z</italic><sub><italic>r</italic></sub> and the leaking rate. In the heterogeneous model case with <italic>x</italic><sub>r</sub>(<italic>t</italic>) and <italic>y</italic><sub>r</sub>(<italic>t</italic>), the complexity on the fast time scales (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1) tended to decrease with the layer depth. The complexity on the slow time scales (&#x003C4;<sub><italic>s</italic></sub> &#x0003D; 10, 20) tended to increase layer by layer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1397915-g0008.tif"/>
</fig>

</sec>

<sec>
<title>3.3 Cross-correlation analysis</title>
<p>The signal transmissions in the dynamics of the reservoir state among layers were evaluated using cross-correlation analysis. <xref ref-type="fig" rid="F9">Figure 9</xref> depicts the dynamics of the reservoir states between adjacent reservoir layers (the <italic>l</italic>-th and <italic>l</italic>&#x0002B;1-th layers): |Corr(<italic>k</italic>)| in the case of the heterogeneous model (spectral radius &#x003C1; &#x0003D; 1.0) for the Mackey&#x02013;Glass (&#x003C4; &#x0003D; 64) task. This setting corresponds to the highest accuracy in <xref ref-type="fig" rid="F3">Figure 3A</xref> for the Mackey&#x02013;Glass (&#x003C4; &#x0003D; 64) task. |Corr(<italic>k</italic>)| was maximized at the positive lag (<italic>k</italic>&#x02265;0) in the specific between layers (&#x00023;3&#x00026;&#x00023;4, &#x00023;4&#x00026;&#x00023;5, &#x00023;5&#x00026;&#x00023;6, &#x00023;6&#x00026;&#x00023;7, &#x00023;8&#x00026;&#x00023;9, &#x00023;9&#x00026;&#x00023;10), i.e., delays in signal transmission occurred from the <italic>l</italic>-th to <italic>l</italic>&#x0002B;1-th layers. To evaluate this tendency against the different tasks used in this study, <xref ref-type="fig" rid="F10">Figure 10</xref> presents the <italic>k</italic>-values where |Corr(<italic>k</italic>)| was maximized for adjacent pair-wise layers. The results demonstrate that a major part of the pair-wise layers in all tasks exhibited positive <italic>k</italic>-values (&#x02265;1); that is, signal transmission delays occurred between adjacent layers. This helps to retain past information. In addition to cross-correlation, it is necessary to evaluate synchronization with delays in systems involving nonlinear dynamics. Therefore, we analyzed the synchronization with delays between the time-series of the reservoir state at the <italic>l</italic>-th and <italic>l</italic>&#x0002B;1-th layers using mutual information under the conditions presented in <xref ref-type="fig" rid="F9">Figure 9</xref> (see Section 3 in <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>). The results indicate that similar to the findings of the cross-correlation, the mutual information peaked at a positive lag (<italic>k</italic>&#x0003E;0), specifically between layers. This indicates that signal transmission delays also occurred from the <italic>l</italic>-th to <italic>l</italic>&#x0002B;1-th layer, considering the nonlinear relationships between the behaviors across layers. In addition, according to the analysis of the output weight matrix <bold>W</bold><sub>out</sub> in the readout, the contributions to the prediction tasks, which were facilitated by these multi-scale behaviors and layer-to-layer delays, were distributed among the layers (see Section 4 in <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>).</p>






<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>Absolute values of cross-correlation for the dynamics of the reservoir states between adjacent reservoir layers (<italic>l</italic>-th and <italic>l</italic>&#x0002B;1-th layers): |Corr(<italic>k</italic>)| in the case with the heterogeneous model (spectral radius &#x003C1; &#x0003D; 1.0) for the Mackey&#x02013;Glass (&#x003C4; &#x0003D; 64) task. This setting corresponds to the highest accuracy in <xref ref-type="fig" rid="F3">Figure 3A</xref> for the Mackey&#x02013;Glass (&#x003C4; &#x0003D; 64) task. Lag <italic>k</italic> where the maximized |Corr(<italic>k</italic>)| was achieved (represented by the red arrow), expresses the signal transmission delay from the <italic>l</italic>-th to <italic>l</italic>&#x0002B;1-th layers. In panels <bold>(C&#x02013;I)</bold>, the peaks are <italic>k</italic>&#x02265;1, indicating that the reservoir dynamics were delayed between layers. <bold>(A)</bold> &#x00023;1 &#x00026; &#x00023;2. <bold>(B)</bold> &#x00023;2 &#x00026; &#x00023;3. <bold>(C)</bold> &#x00023;3 &#x00026; &#x00023;4. <bold>(D)</bold> &#x00023;4 &#x00026; &#x00023;5. <bold>(E)</bold> &#x00023;5 &#x00026; &#x00023;6. <bold>(F)</bold> &#x00023;6 &#x00026; &#x00023;7. <bold>(G)</bold> &#x00023;7 &#x00026; &#x00023;8. <bold>(H)</bold> &#x00023;8 &#x00026; &#x00023;9. <bold>(I)</bold> &#x00023;9 &#x00026; &#x00023;10.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1397915-g0009.tif"/>
</fig>

<fig id="F10" position="float">
<label>Figure 10</label>
<caption><p>Lag <italic>k</italic>-values where |Corr(<italic>k</italic>)| was maximized for adjacent pair-wise layers in the cases with parameter settings (homogeneous/heterogeneous models) to achieve the superior NRMSE in the time-series prediction task (corresponding to the right panel of A in <xref ref-type="fig" rid="F3">Figures 3</xref>&#x02013;<xref ref-type="fig" rid="F5">5</xref>). The values are not plotted when the reservoir dynamics of the <italic>l</italic>-th and <italic>l</italic>&#x0002B;1-th layers were zero-lag synchronized (the peak of correlation was <italic>k</italic> &#x0003D; 0) but plotted when the <italic>l</italic>&#x0002B;1-th layer was delayed against the <italic>l</italic>-th layer (the peak of correlation was <italic>k</italic>&#x02265;1). The presence of between four and six plot points in each subfigure confirms that reservoir dynamics delays occurred in the <italic>l</italic>-th and <italic>l</italic>&#x0002B;1-th layers. <bold>(A)</bold> Mackey-Glass (&#x003C4; = 128), Homogeneous model, leaking rate &#x003B1;= 0.35. <bold>(B)</bold> Mackey-Glass (&#x003C4; = 64), Homogeneous model. <bold>(C)</bold> Lorenz (x-time series), Homogeneous model, leaking rate &#x003B1;= 0.45. <bold>(D)</bold> Lorenz (y-time series), Homogeneous model, leaking rate &#x003B1;= 0.45. <bold>(E)</bold> R&#x000F6;ssler (x-time series), Homogeneous model, leaking rate &#x003B1;= 0.45. <bold>(F)</bold> R&#x000F6;ssler (y-time series), Homogeneous model, leaking rate &#x003B1;= 0.4. <bold>(G)</bold> R&#x000F6;ssler (z-time series), Homogeneous model, leaking rate &#x003B1;= 0.8.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1397915-g0010.tif"/>
</fig>

</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>In this section, we first recapitulate the key findings derived from the results. Our investigation was primarily aimed at understanding the mechanism of the Deep-ESN function enhancement and guidelines by adjusting the leaking rate <italic>a</italic><sup>(<italic>l</italic>)</sup>. In three different experiments, we validated our hypothesis that the evaluation of the dependence of the dynamical response in the multi-layered reservoir on adjusting the leaking rate would provide insights into achieving a guideline for optimizing the hyperparameters of the Deep-ESN. Specifically, the first experiment (time-series prediction tasks) with the homogeneous model showed that the profiles of the MSCE in all tasks and the spectral radii exhibited a <italic>U</italic>-shape against the leaking rate, indicating an optimal leaking rate for each prediction task. The heterogeneous model was effective for the Mackey&#x02013;Glass time series (&#x003C4; &#x0003D; 64), which contains wide multi-temporal-scale components in its dynamics. The second experiment (MSCE analysis) with the homogeneous model indicated that the complexity associated with fast time scales tended to decrease with the leaking rate, while that of the slow time scales tended to vary or remain almost constant according to the number of layers, depending on the task time scale and leaking rate. In the heterogeneous models, the fast time-scale complexity tended to decrease with layer depth, while the slow time-scale complexity tended to increase layer by layer. Finally, the third experiment (cross-correlation analysis) with homogeneous and heterogeneous models demonstrated the layer-to-layer signal transmission delay in all tasks, which helps to retain past information.</p>
<p>We first discuss the reasons for the presence of an optimum leaking rate (see right panel of <xref ref-type="fig" rid="F3">Figures 3</xref>&#x02013;<xref ref-type="fig" rid="F5">5</xref>). In single neural dynamics, as the leaking rate increases, the dynamics of the neuron become faster. Based on this effect, the complexity of fast-scale dynamics in the case with a large leaking rate increases further, especially in deep layers, through the multiple-layered propagation (see the tendency of SampEn when increasing the leaking rate in the case with &#x003C4;<sub><italic>s</italic></sub> &#x0003D; 1 in <xref ref-type="fig" rid="F6">Figures 6</xref>&#x02013;<xref ref-type="fig" rid="F8">8</xref>). This tendency was observed in the homogeneous case with a large leaking rate and in the heterogeneous case. Meanwhile, the complexity at slow time scales exhibited a diverse layer-specific SampEn profile, depending on the leaking rate and task (see the SampEn of &#x003C4;<sub><italic>s</italic></sub> &#x0003D; 10, 20 in <xref ref-type="fig" rid="F6">Figures 6</xref>&#x02013;<xref ref-type="fig" rid="F8">8</xref>). This tendency may be attributed to the complex interactions in layer-to-layer signal propagation. In general, to achieve high ESN performance, the representation of complex desired signals requires the combination of diverse time-scale dynamical responses in the readout (Tanaka et al., <xref ref-type="bibr" rid="B30">2022</xref>). The layer-specific time-scale dynamic response obtained using the Deep-ESN can satisfy this requirement, and this can be achieved by adjusting the leaking rate.</p>
<p>Next, we discuss the structural effectiveness of the Deep-ESN. The hierarchical structure causes delays in the layer-to-layer signal transmission (see <xref ref-type="fig" rid="F9">Figures 9</xref>, <xref ref-type="fig" rid="F10">10</xref>). Consequently, this delay helps to retain the information used in the past, specifically that resembling the queue structure. This characteristic contributes significantly to the high performance (Gallicchio et al., <xref ref-type="bibr" rid="B11">2017</xref>) of the Deep-ESN.</p>
<p>In addition, to adapt the Deep-ESN to real-world data, the characteristics of its performance against time series involving stochastic noise must be considered. In Section 2 of the <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>, the dependence of the NRMSE on the leaking rate is demonstrated against the time-series prediction of the sunspot time series, which involves stochastic behavior. Our results communicate that the estimation performance increases as the leaking rate approaches one. This suggests that suppressing the temporal history effect is essential for accurately modeling time-series data with stochastic noise, although this suppression may compromise the ability to capture long-term behaviors. To address this trade-off, we are currently exploring the integration of attention mechanisms into ESNs (Sakemi et al., <xref ref-type="bibr" rid="B26">2024</xref>), which could further enhance the performance of Deep-ESNs for time-series data involving stochastic noise.</p>
<p>Although this study revealed the existence of an optimal leaking rate, a grid search is still necessary for a concrete set for the leaking rate in accordance with the tasks. This facilitates the need to develop an approach to determine the optimal leaking rate based on the dynamic characteristics used in this study. Moreover, many important benchmark tasks for evaluating deep neural networks have been introduced over the past decade. Notably, Moving Mixed National Institute of Standards and Technology database (MNIST) (Shi et al., <xref ref-type="bibr" rid="B29">2015</xref>), motor imagery datasets (available at <ext-link ext-link-type="uri" xlink:href="https://moabb.neurotechx.com/docs/datasets.html">https://moabb.neurotechx.com/docs/datasets.html</ext-link>), and MLPerf (available at <ext-link ext-link-type="uri" xlink:href="https://mlcommons.org/">https://mlcommons.org/</ext-link>) serve as more contemporary standard benchmarks for deep neural networks. Therefore, it is essential to apply these datasets in addition to the classical prediction tasks described in our study to validate the capability of Deep-ESNs in practical scenarios. These points should be addressed in future studies.</p></sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>In conclusion, through MSCE and cross-correlation analyses, this study has revealed the presence of an optimal leaking rate to represent the complex desired signal and a mechanism to retain past information in the Deep-ESN. Despite some limitations, these findings contribute to establishing optimum design guidelines for setting the hyperparameters of the Deep-ESN.</p></sec>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s10">Supplementary material</xref>, further inquiries can be directed to the corresponding author.</p></sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>SI: Investigation, Methodology, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. SN: Methodology, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. HN: Investigation, Methodology, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. EW: Methodology, Writing &#x02013; review &#x00026; editing. TI: Investigation, Methodology, Writing &#x02013; review &#x00026; editing.</p></sec>
</body>
<back>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This study was supported by a JSPS KAKENHI Grant-in-Aid for Scientific Research (C)(Grant Number JP22K12183) to SN and a JSPS KAKENHI Grant-in-Aid for Transformative Research Areas (A) (Grant Number JP20H05921) to SN. Notably, the funding facilitated the construction of a computational environment essential for our research. The funder was not involved in the study design, the collection, analysis, and interpretation of the data, the writing of this manuscript or the decision to submit it for publication.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>SI was employed by LY Corporation. The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s10">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2024.1397915/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frai.2024.1397915/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.PDF" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Adeleke</surname> <given-names>O. A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Echo-state networks for network traffic prediction,&#x0201D;</article-title> in <source>2019 IEEE 10th Annual Information Technology, Electronics and Mobile Communication Conference (IEMCON)</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>202</fpage>&#x02013;<lpage>206</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Adhikari</surname> <given-names>A.</given-names></name> <name><surname>Sigurdsson</surname> <given-names>T.</given-names></name> <name><surname>Topiwala</surname> <given-names>M. A.</given-names></name> <name><surname>Gordon</surname> <given-names>J. A.</given-names></name></person-group> (<year>2010</year>). <article-title>Cross-correlation of instantaneous amplitudes of field potential oscillations: a straightforward method to estimate the directionality and lag between brain areas</article-title>. <source>J. Neurosci. Methods</source> <volume>191</volume>, <fpage>191</fpage>&#x02013;<lpage>200</lpage>. <pub-id pub-id-type="doi">10.1016/j.jneumeth.2010.06.019</pub-id><pub-id pub-id-type="pmid">20600317</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bai</surname> <given-names>Y.-T.</given-names></name> <name><surname>Jia</surname> <given-names>W.</given-names></name> <name><surname>Jin</surname> <given-names>X.-B.</given-names></name> <name><surname>Su</surname> <given-names>T.-L.</given-names></name> <name><surname>Kong</surname> <given-names>J.-L.</given-names></name> <name><surname>Shi</surname> <given-names>Z.-G.</given-names></name></person-group> (<year>2023</year>). <article-title>Nonstationary time series prediction based on deep echo state network tuned by Bayesian optimization</article-title>. <source>Mathematics</source> <volume>11</volume>:<fpage>1503</fpage>. <pub-id pub-id-type="doi">10.3390/math11061503</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bhandari</surname> <given-names>A.</given-names></name></person-group> (<year>2017</year>). <article-title>Wavelets based multi-scale analysis of select global equity returns</article-title>. <source>Theor. Appl. Econ</source>. <volume>24</volume>:<fpage>613</fpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>S.</given-names></name> <name><surname>Shang</surname> <given-names>P.</given-names></name></person-group> (<year>2020</year>). <article-title>Financial time series analysis using the relation between mpe and mwpe</article-title>. <source>Phys. A Stat. Mech. Appl</source>. <volume>537</volume>:<fpage>122716</fpage>. <pub-id pub-id-type="doi">10.1016/j.physa.2019.122716</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Costa</surname> <given-names>M.</given-names></name> <name><surname>Goldberger</surname> <given-names>A. L.</given-names></name> <name><surname>Peng</surname> <given-names>C.-K.</given-names></name></person-group> (<year>2002</year>). <article-title>Multiscale entropy analysis of complex physiologic time series</article-title>. <source>Phys. Rev. Lett</source>. <volume>89</volume>:<fpage>068102</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevLett.89.068102</pub-id><pub-id pub-id-type="pmid">12190613</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dean</surname> <given-names>R. T.</given-names></name> <name><surname>Dunsmuir</surname> <given-names>W. T.</given-names></name></person-group> (<year>2016</year>). <article-title>Dangers and uses of cross-correlation in analyzing time series in perception, performance, movement, and neuroscience: the importance of constructing transfer function autoregressive models</article-title>. <source>Behav. Res. Methods</source> <volume>48</volume>, <fpage>783</fpage>&#x02013;<lpage>802</lpage>. <pub-id pub-id-type="doi">10.3758/s13428-015-0611-2</pub-id><pub-id pub-id-type="pmid">26100765</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>L.</given-names></name> <name><surname>Yu</surname> <given-names>D.</given-names></name> <name><surname>Platt</surname> <given-names>J.</given-names></name></person-group> (<year>2012</year>). <article-title>&#x0201C;Scalable stacking and learning for building deep architectures,&#x0201D;</article-title> in <source>2012 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2133</fpage>&#x02013;<lpage>2136</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gallicchio</surname> <given-names>C.</given-names></name> <name><surname>Micheli</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Richness of deep echo state network dynamics,&#x0201D;</article-title> in <source>International Work-Conference on Artificial Neural Networks</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>480</fpage>&#x02013;<lpage>491</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gallicchio</surname> <given-names>C.</given-names></name> <name><surname>Micheli</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <source>Deep Reservoir Computing. Reservoir Computing: Theory, Physical Implementations, and Applications</source> (<publisher-loc>Cham</publisher-loc>), <fpage>77</fpage>&#x02013;<lpage>95</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gallicchio</surname> <given-names>C.</given-names></name> <name><surname>Micheli</surname> <given-names>A.</given-names></name> <name><surname>Pedrelli</surname> <given-names>L.</given-names></name></person-group> (<year>2017</year>). <article-title>Deep reservoir computing: a critical experimental analysis</article-title>. <source>Neurocomputing</source> <volume>268</volume>, <fpage>87</fpage>&#x02013;<lpage>99</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2016.12.089</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gallicchio</surname> <given-names>C.</given-names></name> <name><surname>Micheli</surname> <given-names>A.</given-names></name> <name><surname>Pedrelli</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>Comparison between deepesns and gated rnns on multivariate time-series prediction</article-title>. <source>arXiv</source> [preprint]. arXiv:1812.11527. <pub-id pub-id-type="doi">10.48550/arXiv.1812.11527</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Glass</surname> <given-names>L.</given-names></name> <name><surname>Mackey</surname> <given-names>M.</given-names></name></person-group> (<year>2010</year>). <article-title>Mackey-glass equation</article-title>. <source>Scholarpedia</source> <volume>5</volume>:<fpage>6908</fpage>. <pub-id pub-id-type="doi">10.4249/scholarpedia.6908</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Humeau-Heurtier</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>The multiscale entropy algorithm and its variants: a review</article-title>. <source>Entropy</source> <volume>17</volume>, <fpage>3110</fpage>&#x02013;<lpage>3123</lpage>. <pub-id pub-id-type="doi">10.3390/e17053110</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Inoue</surname> <given-names>S.</given-names></name> <name><surname>Nobukawa</surname> <given-names>S.</given-names></name> <name><surname>Nishimura</surname> <given-names>H.</given-names></name> <name><surname>Watanabe</surname> <given-names>E.</given-names></name> <name><surname>Isokawa</surname> <given-names>T.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Mechanism for enhancement of functionality in deep echo state network by optimizing leaking rate,&#x0201D;</article-title> in <source>2023 International Conference on Emerging Techniques in Computational Intelligence (ICETCI)</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>85</fpage>&#x02013;<lpage>90</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jaeger</surname> <given-names>H.</given-names></name></person-group> (<year>2001</year>). <source>The &#x0201C;Echo State&#x0201D; Approach to Analysing and Training Recurrent Neural Networks-With an Erratum Note</source>. <publisher-loc>Bonn</publisher-loc>: <publisher-name>German National Research Center for Information Technology GMD Technical Report 148, 13</publisher-name>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jaeger</surname> <given-names>H.</given-names></name></person-group> (<year>2007</year>). <article-title>Echo state network</article-title>. <source>Scholarpedia</source> <volume>2</volume>:<fpage>2330</fpage>. <pub-id pub-id-type="doi">10.4249/scholarpedia.2330</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jaeger</surname> <given-names>H.</given-names></name> <name><surname>Luko&#x00161;evi&#x0010D;ius</surname> <given-names>M.</given-names></name> <name><surname>Popovici</surname> <given-names>D.</given-names></name> <name><surname>Siewert</surname> <given-names>U.</given-names></name></person-group> (<year>2007</year>). <article-title>Optimization and applications of echo state networks with leaky-integrator neurons</article-title>. <source>Neural Netw</source>. <volume>20</volume>, <fpage>335</fpage>&#x02013;<lpage>352</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2007.04.016</pub-id><pub-id pub-id-type="pmid">17517495</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kanda</surname> <given-names>K.</given-names></name> <name><surname>Nobukawa</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Feature extraction mechanism for each layer of deep echo state network,&#x0201D;</article-title> in <source>2022 International Conference on Emerging Techniques in Computational Intelligence (ICETCI)</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>65</fpage>&#x02013;<lpage>70</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Long</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name></person-group> (<year>2019</year>). <article-title>Evolving deep echo state networks for intelligent fault diagnosis</article-title>. <source>IEEE Transact. Ind. Inf</source>. <volume>16</volume>, <fpage>4928</fpage>&#x02013;<lpage>4937</lpage>. <pub-id pub-id-type="doi">10.1109/TII.2019.2938884</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luko&#x00161;evi&#x0010D;ius</surname> <given-names>M.</given-names></name> <name><surname>Jaeger</surname> <given-names>H.</given-names></name></person-group> (<year>2009</year>). <article-title>Reservoir computing approaches to recurrent neural network training</article-title>. <source>Comp. Sci. Rev</source>. <volume>3</volume>, <fpage>127</fpage>&#x02013;<lpage>149</lpage>. <pub-id pub-id-type="doi">10.1016/j.cosrev.2009.03.005</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Luko&#x00161;evi&#x0010D;ius</surname> <given-names>M.</given-names></name> <name><surname>Uselis</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Efficient cross-validation of echo state networks,&#x0201D;</article-title> in <source>Artificial Neural Networks and Machine Learning-ICANN 2019: Workshop and Special Sessions: 28th International Conference on Artificial Neural Networks, Munich, Germany, September 17-19, 2019, Proceedings 28</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>121</fpage>&#x02013;<lpage>133</lpage>.<pub-id pub-id-type="pmid">38901093</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Malik</surname> <given-names>Z. K.</given-names></name> <name><surname>Hussain</surname> <given-names>A.</given-names></name> <name><surname>Wu</surname> <given-names>Q. J.</given-names></name></person-group> (<year>2016</year>). <article-title>Multilayered echo state machine: a novel architecture and algorithm</article-title>. <source>IEEE Trans. Cybern</source>. <volume>47</volume>, <fpage>946</fpage>&#x02013;<lpage>959</lpage>. <pub-id pub-id-type="doi">10.1109/TCYB.2016.2533545</pub-id><pub-id pub-id-type="pmid">27337730</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Manneville</surname> <given-names>P.</given-names></name> <name><surname>Pomeau</surname> <given-names>Y.</given-names></name></person-group> (<year>1979</year>). <article-title>Intermittency and the lorenz model</article-title>. <source>Phys. Lett. A</source> <volume>75</volume>, <fpage>1</fpage>&#x02013;<lpage>2</lpage>. <pub-id pub-id-type="doi">10.1016/0375-9601(79)90255-X</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>R&#x000F6;ssler</surname> <given-names>O. E.</given-names></name></person-group> (<year>1983</year>). <article-title>The chaotic hierarchy</article-title>. <source>Zeitschrift Naturforschung A</source> <volume>38</volume>, <fpage>788</fpage>&#x02013;<lpage>801</lpage>. <pub-id pub-id-type="doi">10.1515/zna-1983-0714</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sakemi</surname> <given-names>Y.</given-names></name> <name><surname>Nobukawa</surname> <given-names>S.</given-names></name> <name><surname>Matsuki</surname> <given-names>T.</given-names></name> <name><surname>Morie</surname> <given-names>T.</given-names></name> <name><surname>Aihara</surname> <given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>Learning reservoir dynamics with temporal self-modulation</article-title>. <source>Commun. Phys</source>. <volume>7</volume>:<fpage>29</fpage>. <pub-id pub-id-type="doi">10.1038/s42005-023-01500-w</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Salehinejad</surname> <given-names>H.</given-names></name> <name><surname>Sankar</surname> <given-names>S.</given-names></name> <name><surname>Barfett</surname> <given-names>J.</given-names></name> <name><surname>Colak</surname> <given-names>E.</given-names></name> <name><surname>Valaee</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>Recent advances in recurrent neural networks</article-title>. <source>arXiv</source> [preprint]. arXiv:1801.01078. <pub-id pub-id-type="doi">10.48550/arXiv.1801.01078</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Schrauwen</surname> <given-names>B.</given-names></name> <name><surname>Defour</surname> <given-names>J.</given-names></name> <name><surname>Verstraeten</surname> <given-names>D.</given-names></name> <name><surname>Van Campenhout</surname> <given-names>J.</given-names></name></person-group> (<year>2007</year>). <article-title>&#x0201C;The introduction of time-scales in reservoir computing, applied to isolated digits recognition,&#x0201D;</article-title> in <source>Artificial Neural Networks-ICANN 2007: 17th International Conference, Porto, Portugal, September 9-13, 2007, Proceedings, Part I 17</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>471</fpage>&#x02013;<lpage>479</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Yeung</surname> <given-names>D.-Y.</given-names></name> <name><surname>Wong</surname> <given-names>W.-K.</given-names></name> <name><surname>Woo</surname> <given-names>W.-C.</given-names></name></person-group> (<year>2015</year>). <article-title>Convolutional lstm network: a machine learning approach for precipitation nowcasting</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>28</volume>, <fpage>802</fpage>&#x02013;<lpage>810</lpage>. <pub-id pub-id-type="doi">10.5555/2969239.2969329</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tanaka</surname> <given-names>G.</given-names></name> <name><surname>Matsumori</surname> <given-names>T.</given-names></name> <name><surname>Yoshida</surname> <given-names>H.</given-names></name> <name><surname>Aihara</surname> <given-names>K.</given-names></name></person-group> (<year>2022</year>). <article-title>Reservoir computing with diverse timescales for prediction of multiscale dynamics</article-title>. <source>Phys. Rev. Res</source>. <volume>4</volume>:<fpage>L032014</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevResearch.4.L032014</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tanaka</surname> <given-names>G.</given-names></name> <name><surname>Yamane</surname> <given-names>T.</given-names></name> <name><surname>H&#x000E9;roux</surname> <given-names>J. B.</given-names></name> <name><surname>Nakane</surname> <given-names>R.</given-names></name> <name><surname>Kanazawa</surname> <given-names>N.</given-names></name> <name><surname>Takeda</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Recent advances in physical reservoir computing: a review</article-title>. <source>Neural Netw</source>. <volume>115</volume>, <fpage>100</fpage>&#x02013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2019.03.005</pub-id><pub-id pub-id-type="pmid">30981085</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tchakoucht</surname> <given-names>T. A.</given-names></name> <name><surname>Ezziyyani</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>Multilayered echo-state machine: a novel architecture for efficient intrusion detection</article-title>. <source>IEEE Access</source> <volume>6</volume>, <fpage>72458</fpage>&#x02013;<lpage>72468</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2018.2867345</pub-id><pub-id pub-id-type="pmid">27337730</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Venkatasubramanian</surname> <given-names>V.</given-names></name> <name><surname>Schattler</surname> <given-names>H.</given-names></name> <name><surname>Zaborsky</surname> <given-names>J.</given-names></name></person-group> (<year>1995</year>). <article-title>Dynamics of large constrained nonlinear systems-a taxonomy theory [power system stability]</article-title>. <source>Proc. IEEE</source> <volume>83</volume>, <fpage>1530</fpage>&#x02013;<lpage>1561</lpage>. <pub-id pub-id-type="doi">10.1109/5.481633</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Viehweg</surname> <given-names>J.</given-names></name> <name><surname>Worthmann</surname> <given-names>K.</given-names></name> <name><surname>M&#x000E4;der</surname> <given-names>P.</given-names></name></person-group> (<year>2023</year>). <article-title>Parameterizing echo state networks for multi-step time series prediction</article-title>. <source>Neurocomputing</source> <volume>522</volume>, <fpage>214</fpage>&#x02013;<lpage>228</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2022.11.044</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Werbos</surname> <given-names>P. J.</given-names></name></person-group> (<year>1990</year>). <article-title>Backpropagation through time: what it does and how to do it</article-title>. <source>Proc. IEEE</source> <volume>78</volume>, <fpage>1550</fpage>&#x02013;<lpage>1560</lpage>. <pub-id pub-id-type="doi">10.1109/5.58337</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Williams</surname> <given-names>R. J.</given-names></name> <name><surname>Zipser</surname> <given-names>D.</given-names></name></person-group> (<year>1989</year>). <article-title>A learning algorithm for continually running fully recurrent neural networks</article-title>. <source>Neural Comput</source>. <volume>1</volume>, <fpage>270</fpage>&#x02013;<lpage>280</lpage>. <pub-id pub-id-type="doi">10.1162/neco.1989.1.2.270</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>B.</given-names></name> <name><surname>He</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Dynamics and complexity analysis of the conformable fractional-order two-machine interconnected power system</article-title>. <source>Math. Methods Appl. Sci</source>. <volume>44</volume>, <fpage>2439</fpage>&#x02013;<lpage>2454</lpage>. <pub-id pub-id-type="doi">10.1002/mma.5937</pub-id></citation>
</ref>
</ref-list>
</back>
</article>