<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2024.1362510</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>An analytical approach for unsupervised learning rate estimation using rectified linear units</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Chaoxiang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Golovko</surname> <given-names>Vladimir</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/244604/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Kroshchanka</surname> <given-names>Aliaksandr</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Mikhno</surname> <given-names>Egor</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Chodyka</surname> <given-names>Marta</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2616456/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lichograj</surname> <given-names>Piotr</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1937174/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Information Science and Technology, Zhejiang Shuren University</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>International Science and Technology Cooperation Base of Zhejiang Province: Remote Sensing Image Processing and Application</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Institute of Traditional Chinese Medicine Artificial Intelligence Zhejiang Shuren University</institution>, <addr-line>Hangzhou</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Computer Science, John Paul II University in Biala Podlaska</institution>, <addr-line>Biala Podlaska</addr-line>, <country>Poland</country></aff>
<aff id="aff5"><sup>5</sup><institution>Intelligent Information Technologies Department, Brest State Technical University</institution>, <addr-line>Brest</addr-line>, <country>Belarus</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: Anguo Zhang, University of Macau, China</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: Omid Memarian Sorkhabi, University College Dublin, Ireland</p>
<p>Junyi Wu, Fuzhou University, China</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Vladimir Golovko, <email>vladimir.golovko@gmail.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>08</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>18</volume>
<elocation-id>1362510</elocation-id>
<history>
<date date-type="received">
<day>28</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>28</day>
<month>02</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Chen, Golovko, Kroshchanka, Mikhno, Chodyka and Lichograj.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Chen, Golovko, Kroshchanka, Mikhno, Chodyka and Lichograj</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Unsupervised learning based on restricted Boltzmann machine or autoencoders has become an important research domain in the area of neural networks. In this paper mathematical expressions to adaptive learning step calculation for RBM with ReLU transfer function are proposed. As a result, we can automatically estimate the step size that minimizes the loss function of the neural network and correspondingly update the learning step in every iteration. We give a theoretical justification for the proposed adaptive learning rate approach, which is based on the steepest descent method. The proposed technique for adaptive learning rate estimation is compared with the existing constant step and Adam methods in terms of generalization ability and loss function. We demonstrate that the proposed approach provides better performance.</p>
</abstract>
<kwd-group id="ae_14">
<kwd>adaptive training step</kwd>
<kwd>RBM</kwd>
<kwd>deep learning</kwd>
<kwd>unsupervised learning</kwd>
<kwd>ReLU</kwd>
<kwd>activation function</kwd>
<kwd>Adam</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="8"/>
<equation-count count="71"/>
<ref-count count="48"/>
<page-count count="14"/>
<word-count count="10124"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Decision Neuroscience</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>During recent years many papers have been devoted to the study of restricted Boltzmann machines (RBM) and more generally to that of deep learning, because it is a breakthrough approach in the field of artificial intelligence (<xref ref-type="bibr" rid="ref22">Hinton, 2002</xref>, <xref ref-type="bibr" rid="ref23">2010</xref>; <xref ref-type="bibr" rid="ref25">Hinton et al., 2006</xref>; <xref ref-type="bibr" rid="ref26">Hinton and Salakhutdinov, 2006</xref>; <xref ref-type="bibr" rid="ref36">Nair and Hinton, 2010</xref>; <xref ref-type="bibr" rid="ref29">Krizhevsky et al., 2012</xref>; <xref ref-type="bibr" rid="ref32">LeCun et al., 2015</xref>). Deep learning has been developing very quickly in the last decade. As a result, various successful applications of deep learning have been proposed in speech recognition, computer vision, natural language processing, data visualization, etc. (<xref ref-type="bibr" rid="ref6">Bengio et al., 2007</xref>, <xref ref-type="bibr" rid="ref5">2013</xref>, <xref ref-type="bibr" rid="ref7">2021</xref>; <xref ref-type="bibr" rid="ref4">Bengio, 2009</xref>; <xref ref-type="bibr" rid="ref31">Larochelle et al., 2009</xref>; <xref ref-type="bibr" rid="ref14">Erhan et al., 2010</xref>; <xref ref-type="bibr" rid="ref17">Golovko et al., 2010</xref>; <xref ref-type="bibr" rid="ref15">Glorot et al., 2011</xref>; <xref ref-type="bibr" rid="ref35">Mikolov et al., 2011</xref>; <xref ref-type="bibr" rid="ref24">Hinton et al., 2012</xref>; <xref ref-type="bibr" rid="ref33">Madani et al., 2018</xref>; <xref ref-type="bibr" rid="ref30">Lamb et al., 2022</xref>; <xref ref-type="bibr" rid="ref46">Verma et al., 2022</xref>; <xref ref-type="bibr" rid="ref1">Aguilera et al., 2023</xref>; <xref ref-type="bibr" rid="ref34">Menezes et al., 2023</xref>; <xref ref-type="bibr" rid="ref10">Chen et al., 2024</xref>).</p>
<p>One of the major and important problems in this domain is the selection of suitable hyperparameters values to achieve significant performance of a neural network. Among these parameters, the learning rate is of great importance because it has a significant impact on the training efficiency of the neural network (<xref ref-type="bibr" rid="ref16">Golovko, 2003</xref>; <xref ref-type="bibr" rid="ref11">Cho et al., 2011</xref>; <xref ref-type="bibr" rid="ref13">Duchi et al., 2011</xref>; <xref ref-type="bibr" rid="ref28">Krizhevsky and Hinton, 2012</xref>; <xref ref-type="bibr" rid="ref48">Zeiler, 2012</xref>; <xref ref-type="bibr" rid="ref41">Schaul et al., 2013</xref>; <xref ref-type="bibr" rid="ref27">Kingma and Ba, 2014</xref>; <xref ref-type="bibr" rid="ref40">Ruder, 2016</xref>; <xref ref-type="bibr" rid="ref39">Pouyanfar and Chen, 2017</xref>; <xref ref-type="bibr" rid="ref43">Smith, 2017</xref>; <xref ref-type="bibr" rid="ref3">Baydin et al., 2018</xref>; <xref ref-type="bibr" rid="ref44">Takase et al., 2018</xref>; <xref ref-type="bibr" rid="ref2">Arpit and Bengio, 2019</xref>; <xref ref-type="bibr" rid="ref45">Vaswani et al., 2019</xref>; <xref ref-type="bibr" rid="ref38">Pesme et al., 2020</xref>; <xref ref-type="bibr" rid="ref8">Carvalho et al., 2021</xref>; <xref ref-type="bibr" rid="ref37">Nakamura et al., 2021</xref>; <xref ref-type="bibr" rid="ref9">Chen et al., 2022</xref>; <xref ref-type="bibr" rid="ref12">Defazio et al., 2023</xref>; <xref ref-type="bibr" rid="ref20">Golovko et al., 2023</xref>; <xref ref-type="bibr" rid="ref47">Wang et al., 2023</xref>). The choice of an appropriate learning rate controls how well the neural network adapts to the problem being solved and achieves a suitable minimum of the loss function. So, for instance, for many applications the learning rate has to be manually and carefully chosen, because depending on this parameter the learning process can be divergent or convergent. Therefore, to avoid these problems, learning step should be defined and modified automatically during neural network learning.</p>
<p>The neural networks community has been concerned with this problem for many years, and currently there are only partial solutions to selecting an appropriate learning rate. This situation gives rise to the question of how we can obtain analytical expressions for learning rate calculation. This question is addressed in the present paper. As a result, an analytical approach to estimate the value of the learning step has been proposed, based on the steepest descent approach. The proposed approach capables to automatically defining and adjusting the learning rate during the training of a neural network.</p>
<p>In our previous work (<xref ref-type="bibr" rid="ref20">Golovko et al., 2023</xref>), we proposed an approach to estimate the learning rate of a single-layer perceptron with a rectified linear unit activation function (ReLU). The present article focuses on an adaptive learning step (ATS) for RBM with a ReLU. It is the simplest activation function, which is a piecewise linear function consisting of two straight lines. ReLU is not a saturated activation function with unlimited output, unlike other activation functions. It has been noted in existing literature that using a ReLU network generally improves performance (<xref ref-type="bibr" rid="ref45">Vaswani et al., 2019</xref>; <xref ref-type="bibr" rid="ref47">Wang et al., 2023</xref>). As stated in the article (<xref ref-type="bibr" rid="ref36">Nair and Hinton, 2010</xref>) rectified linear units can improve RBM. As well is known a RBM can be applied for deep neural networks learning (<xref ref-type="bibr" rid="ref32">LeCun et al., 2015</xref>). The conventional approach to RBM learning usually uses constant or empirically varying learning step (<xref ref-type="bibr" rid="ref11">Cho et al., 2011</xref>). Currently, there are no analytical expressions to estimate the learning rate, which can be automatically defining and adjusting the learning rate during the training of a RBM network. As a rule, there are only empirical and heuristic approaches to set learning rate.</p>
<p>Therefore, in this paper we investigate the calculation of adaptive learning rate for a RBM, which is based on the steepest descent technique (<xref ref-type="bibr" rid="ref21">Golovko et al., 2000</xref>, <xref ref-type="bibr" rid="ref20">2023</xref>; <xref ref-type="bibr" rid="ref16">Golovko, 2003</xref>). This approach is based on minimizing the loss function to calculate the adaptive learning step. Since derivation an accurate analytical expression for estimating the learning rate using steepest descent approach is a very difficult task, most scientists use the steepest descent method together with the line search approach. However, as we will show in this article, it is possible to derive exact expressions for the RBM learning rate using the ReLU activation function. The adaptive learning rate approach permits to compute the learning step at each time. An advantage of the proposed approach is that we can automatically estimate a specific learning rate value for each batch or each example from the training data set.</p>
<p>Further, we perform stacking ReLU RBM into a deep neural network. As a result, we can train deep neural networks using unsupervised and SGD techniques.</p>
<p>The major contribution of this paper is novel mathematical expressions for adaptive learning rate calculation, if we use RBM with ReLU transfer function. The proposed approach is based on steepest descent technique and allows to estimate the ATS at each iteration of the learning algorithm. We have shown, using a set of experiments, that the proposed adaptive learning rate can improve performance with respect to learning quality and generalization ability.</p>
<p>In the present study we proceed as follows. Section 2 introduces the related work in this area. In Section 3 we consider different representations of RBM. Section 4 deals with learning rules for RBM with ReLU. In section 5 we propose the adaptive learning step calculation for RBM. Section 6 demonstrates the results of experiments, and finally we give our conclusion.</p>
</sec>
<sec id="sec2">
<label>2</label>
<title>Related work</title>
<p>In the following, a brief overview of related works in this area is presented. It is well known that there are the two principal techniques for learning of deep neural networks (DNN): learning with pretraining using a greedy layer wise approach and stochastic gradient descent approach (SGD), including its various modifications. If we do not use pretraining of DNN, then it is necessary to use a rectified linear unit (ReLU) transfer function, because of the vanishing gradient problem (<xref ref-type="bibr" rid="ref32">LeCun et al., 2015</xref>).</p>
<p>RBM can be used as building blocks for deep neural networks, where every layer of neural network is trained as RBM in an unsupervised manner (<xref ref-type="bibr" rid="ref22">Hinton, 2002</xref>, <xref ref-type="bibr" rid="ref23">2010</xref>; <xref ref-type="bibr" rid="ref25">Hinton et al., 2006</xref>; <xref ref-type="bibr" rid="ref26">Hinton and Salakhutdinov, 2006</xref>; <xref ref-type="bibr" rid="ref36">Nair and Hinton, 2010</xref>). By stacking RBMs in this way, one can obtain a suitable initialization of a deep neural network for further training using a backpropagation algorithm.</p>
<p>For smaller data sets, unsupervised pretraining helps to prevent overfitting (<xref ref-type="bibr" rid="ref32">LeCun et al., 2015</xref>). As stated in paper (<xref ref-type="bibr" rid="ref32">LeCun et al., 2015</xref>): &#x201C;Although at present the supervised training with ReLU is used mainly for deep neural networks learning, we expect unsupervised learning to become far more important in the longer term. Human and animal learning is largely unsupervised: we discover the structure of the world by observing it, not by being told the name of every object.&#x201D; Consequently, unsupervised learning is of great importance. Therefore, we consider in this work the different representations of RBM and study estimation of an adaptive learning rate.</p>
<p>Currently the most methods for learning rate estimation are oriented to the SGD approach (<xref ref-type="bibr" rid="ref13">Duchi et al., 2011</xref>; <xref ref-type="bibr" rid="ref48">Zeiler, 2012</xref>; <xref ref-type="bibr" rid="ref41">Schaul et al., 2013</xref>; <xref ref-type="bibr" rid="ref27">Kingma and Ba, 2014</xref>; <xref ref-type="bibr" rid="ref40">Ruder, 2016</xref>; <xref ref-type="bibr" rid="ref39">Pouyanfar and Chen, 2017</xref>; <xref ref-type="bibr" rid="ref43">Smith, 2017</xref>; <xref ref-type="bibr" rid="ref3">Baydin et al., 2018</xref>; <xref ref-type="bibr" rid="ref44">Takase et al., 2018</xref>; <xref ref-type="bibr" rid="ref45">Vaswani et al., 2019</xref>; <xref ref-type="bibr" rid="ref37">Nakamura et al., 2021</xref>; <xref ref-type="bibr" rid="ref9">Chen et al., 2022</xref>; <xref ref-type="bibr" rid="ref12">Defazio et al., 2023</xref>; <xref ref-type="bibr" rid="ref47">Wang et al., 2023</xref>). If the SGD approach is used, then, as a rule, an initial learning rate is selected manually, and further during the learning, the training rate is decreased over time, using different rules. We have not found any works as concerns analytical expressions for the learning rate estimation. There are various approaches to learning rate estimation using different versions of SGD. Let us consider these approaches shortly.</p>
<p>Existing works related to learning rate selection are based mostly on learning rate schedule or line search approach. So, for instance the estimation of adaptive learning rate using line search approach is proposed in <xref ref-type="bibr" rid="ref45">Vaswani et al. (2019)</xref> and <xref ref-type="bibr" rid="ref47">Wang et al. (2023)</xref>. As mentioned earlier, as a rule, the line search approach is used in conjunction with the steepest descent technique. However, such an approach is computationally expensive and time consuming. Furthermore, as will be shown in this paper, it is possible to obtain for RBM with ReLU precise expressions for the learning rate instead of using line search. Learning rate scheduling is a very popular approach and is used in various gradient descent optimization algorithms, namely, Adagrad, Adadelta, RMSprop, and Adam. The primary shortcoming associated with learning rate schedules is their dependence on predefined initial learning rate.</p>
<p>So, for instance, the Adagrad method (<xref ref-type="bibr" rid="ref13">Duchi et al., 2011</xref>) divides the learning rate at each step by the norm of all previous gradients. The other approaches, such as Adadelta and Adam are based on Adagrad and as a result the learning rate decreases during training (<xref ref-type="bibr" rid="ref27">Kingma and Ba, 2014</xref>; <xref ref-type="bibr" rid="ref40">Ruder, 2016</xref>). In <xref ref-type="bibr" rid="ref38">Pesme et al. (2020)</xref>, the optimization process of SGD is divided into two stages: transient stage and stationary stage. It should be noted that the learning step is reduced during the stationary phase. In <xref ref-type="bibr" rid="ref43">Smith (2017)</xref>, scheduling learning rate is performed for each iteration. In <xref ref-type="bibr" rid="ref3">Baydin et al. (2018)</xref>, the hypergradient descent approach is proposed in order to find appropriate learning step. In <xref ref-type="bibr" rid="ref37">Nakamura et al. (2021)</xref>, ATS technique is proposed, which is based on a combination of reducing and increasing the learning rate.</p>
<p>As regards analytical learning rate at the pretraining stage, we have not found any works as concerns the learning rate estimation. Substantially, all known approaches are based again not on analytical expressions for calculating the learning rate, but on empirical approaches and the policy of changing the learning step. So, for instance, in <xref ref-type="bibr" rid="ref11">Cho et al. (2011)</xref> for RBM is proposed an approach to automatically adjust the learning rate by maximizing a local likelihood estimate. However, as a result, the learning rate is chosen based on the previous learning rate and a small constant, that leads again in manual selection of the initial parameters.</p>
<p>In this paper we propose to use steepest descent approach to derive learning rate. Such learning rate can only be obtained for linear and ReLU activation functions. When using the sigmoid activation function, we can only receive approximate expressions for the learning rate using the Taylor series expansion (<xref ref-type="bibr" rid="ref21">Golovko et al., 2000</xref>; <xref ref-type="bibr" rid="ref16">Golovko, 2003</xref>). Since this is a very complicated problem, as mentioned before, most of the scientists use the steepest descent method together with the line search approach.</p>
<p>Our previous work (<xref ref-type="bibr" rid="ref20">Golovko et al., 2023</xref>) reported an adaptive learning rate for a single-layer perceptron with a ReLU activation function. Let us consider the simplest neural network, namely single layer perceptron (SLP). In the case of a single-layer perceptron with ReLU activation function, the expressions for calculating the adaptive learning step was obtained for the first time in the work (<xref ref-type="bibr" rid="ref20">Golovko et al., 2023</xref>) based on the proof of the following theorems:</p>
<p><bold>Theorem 1:</bold> For a single-layer perceptron with a ReLU activation function in the case of online learning, the value of the adaptive learning step is calculated based on the following expression <xref ref-type="disp-formula" rid="EQ1">Eq. (1)</xref>:</p>
<disp-formula id="EQ1">
<label>(1)</label>
<mml:math id="M1">
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<disp-formula id="E1">
<mml:math id="M2">
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mtext>,</mml:mtext>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math id="M3">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="center">
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>0.</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</inline-formula></p>
<p>Here <inline-formula>
<mml:math id="M4">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M5">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> denotes corresponding slopes of the ReLU function; <inline-formula>
<mml:math id="M6">
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:math>
</inline-formula> is desired output for j-th unit; n and m denotes the number of input and output unit, <inline-formula>
<mml:math id="M7">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M8">
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:math>
</inline-formula> are weighted sum and output of the j-th unit.</p>
<p>It should be noted, that <inline-formula>
<mml:math id="M9">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2260;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> and 0 &#x003C; r<sub>2</sub> &#x003C; 1.</p>
<p><bold>Theorem 2:</bold> For a single-layer perceptron with a ReLU activation function in the case of batch learning, the value of the adaptive learning step is calculated based on the following expression <xref ref-type="disp-formula" rid="EQ2">Eq. (2)</xref>:</p>
<disp-formula id="EQ2">
<label>(2)</label>
<mml:math id="M10">
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:msubsup>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:msubsup>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<disp-formula id="E2">
<mml:math id="M11">
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mtext>,</mml:mtext>
</mml:math>
</disp-formula>
<p>where<inline-formula>
<mml:math id="M12">
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> is batch size and <inline-formula>
<mml:math id="M13">
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="center">
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>0.</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</inline-formula></p>
<p>As stated in <xref ref-type="bibr" rid="ref20">Golovko et al. (2023)</xref>, the above expressions <xref ref-type="disp-formula" rid="EQ1">Eqs. (1</xref>, <xref ref-type="disp-formula" rid="EQ2">2)</xref> can significantly increase the learning quality of a single-layer perceptron and achieve an optimal solution to the problem. The proposed approach was generalized to unsupervised pretraining of deep neural network (<xref ref-type="bibr" rid="ref20">Golovko et al., 2023</xref>), using autoencoder method. The primary goal of the present work is to obtain the analytical expressions to learning rate estimation for restricted Boltzmann machine with ReLU activation function.</p>
</sec>
<sec id="sec3">
<label>3</label>
<title>Restricted Boltzmann machine</title>
<p>In this section we consider different representation of RBM from structure and learning point of view.</p>
<p>Let us consider a conventional restricted Boltzmann machine (<xref ref-type="bibr" rid="ref23">Hinton, 2010</xref>), which has bipartite structure consisting of two layers: a visible layer containing n units and hidden layer containing m units (<xref ref-type="fig" rid="fig1">Figure 1</xref>).</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Restricted Boltzmann machine.</p>
</caption>
<graphic xlink:href="fnins-18-1362510-g001.tif"/>
</fig>
<p>In the RBM structure, each neuron in visible layer is connected to all the units in the hidden layer, using bidirectional weights W. RBM can be used as main building blocks for deep neural networks (<xref ref-type="bibr" rid="ref22">Hinton, 2002</xref>, <xref ref-type="bibr" rid="ref23">2010</xref>; <xref ref-type="bibr" rid="ref25">Hinton et al., 2006</xref>; <xref ref-type="bibr" rid="ref36">Nair and Hinton, 2010</xref>). Usually the states of visible and hidden units are defined using a probabilistic version of the sigmoid activation function according to <xref ref-type="disp-formula" rid="EQ3">Eqs. (3</xref>, <xref ref-type="disp-formula" rid="EQ4">4)</xref>:</p>
<disp-formula id="EQ3">
<label>(3)</label>
<mml:math id="M14">
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo stretchy="true">|</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<disp-formula id="EQ4">
<label>(4)</label>
<mml:math id="M15">
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">|</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<p>It should be noted that the variables at the hidden layer are independent given the state of the visible units, and vice versa as shown in expression <xref ref-type="disp-formula" rid="EQ5">Eq. (5)</xref>:</p>
<disp-formula id="EQ5">
<label>(5)</label>
<mml:math id="M16">
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo stretchy="true">|</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x220F;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">|</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="E4">
<mml:math id="M17">
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">|</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x220F;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo stretchy="true">|</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>The hidden units of the RBM can be interpreted as feature detectors which capture the regularities of the input data. The traditional way of getting the training rule is to maximize the function of log-likelihood of the input data distribution P(x). In other words, it is necessary to reproduce the distribution of input data as closely as possible using the states of hidden units. The main properties of conventional RBM are the following: symmetric weights in the hidden and visible layers; Gibbs sampling during the training and stochastic neurons. Next, we will consider a RBM that is characterized only by the first two properties, and the neurons are not stochastic.</p>
<p>Let us consider unfolded representation of the RBM using three layers (visible, hidden and visible; <xref ref-type="bibr" rid="ref19">Golovko et al., 2015</xref>, <xref ref-type="bibr" rid="ref18">2016</xref>) as shown in <xref ref-type="fig" rid="fig2">Figure 2</xref>. Such a representation of RBM is equivalent to PCA or autoencoder neural network, where the hidden and last visible layer is, respectively, compression and reconstruction layer.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Unfolded representation of RBM.</p>
</caption>
<graphic xlink:href="fnins-18-1362510-g002.tif"/>
</fig>
<p>Let us consider the Gibbs sampling using CD-k. In this case we can represent Gibbs sampling for above structure as shown in <xref ref-type="fig" rid="fig3">Figure 3</xref>.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Gibbs sampling.</p>
</caption>
<graphic xlink:href="fnins-18-1362510-g003.tif"/>
</fig>
<p>Next, we will consider Gibbs sampling for CD-1. Let x(0) is the input data, that enter at the visible layer at time 0. Then the output of the hidden layer is defined as follows <xref ref-type="disp-formula" rid="EQ6">Eqs. (6</xref>, <xref ref-type="disp-formula" rid="EQ7">7)</xref>:</p>
<disp-formula id="EQ6">
<label>(6)</label>
<mml:math id="M18">
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mi>F</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ7">
<label>(7)</label>
<mml:math id="M19">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mi>i</mml:mi>
</mml:munder>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<p>The reconstruction layer reproducts the data from the hidden layer. As a result we can obtain x(1) at time 1 using <xref ref-type="disp-formula" rid="EQ8">Eqs. (8</xref>, <xref ref-type="disp-formula" rid="EQ9">9)</xref>:</p>
<disp-formula id="EQ8">
<label>(8)</label>
<mml:math id="M20">
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mi>F</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ9">
<label>(9)</label>
<mml:math id="M21">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mi>j</mml:mi>
</mml:munder>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<p>After this, x(1) enters the visible layer and we can obtain the output of the hidden layer the following way <xref ref-type="disp-formula" rid="EQ10">Eqs. (10</xref>, <xref ref-type="disp-formula" rid="EQ11">11)</xref>:</p>
<disp-formula id="EQ10">
<label>(10)</label>
<mml:math id="M22">
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mi>F</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ11">
<label>(11)</label>
<mml:math id="M23">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mi>i</mml:mi>
</mml:munder>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<p>As mentioned before the conventional approach of getting the training rule is to maximize the function of log-likelihood of the input data distribution. In <xref ref-type="bibr" rid="ref19">Golovko et al. (2015</xref>, <xref ref-type="bibr" rid="ref18">2016)</xref>, we have proposed an alternative approach in order to obtain RBM learning rule, which is based on the minimization of mean square error (MSE). As stated in <xref ref-type="bibr" rid="ref18">Golovko et al. (2016)</xref> the primary goal of training RBM is to minimize the reconstruction mean squared error (MSE) in the hidden and visible layers simultaneously. The MSE in the hidden layer is proportional to the difference between the states of the hidden units at the various time steps. Then in case of CD-1 the MSE in the hidden layer is defined as shown in expression <xref ref-type="disp-formula" rid="EQ12">Eq. (12)</xref>:</p>
<disp-formula id="EQ12">
<label>(12)</label>
<mml:math id="M24">
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>L</mml:mi>
</mml:munderover>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:math>
</disp-formula>
<p>Similarly, the MSE in the inverse layer is proportional to the difference between the states of the inverse units at the various time steps <xref ref-type="disp-formula" rid="EQ13">Eq. (13)</xref>:</p>
<disp-formula id="EQ13">
<label>(13)</label>
<mml:math id="M25">
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>L</mml:mi>
</mml:munderover>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mo stretchy="true">(</mml:mo>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:math>
</disp-formula>
<p>where L is the number of training patterns.</p>
<p>Then the main purpose of the training RBM is to minimize the total mean squared error (MSE), which is defined as the sum of errors <xref ref-type="disp-formula" rid="EQ14">Eq. (14)</xref>:</p>
<disp-formula id="EQ14">
<label>(14)</label>
<mml:math id="M26">
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>The following theorem is proved in <xref ref-type="bibr" rid="ref18">Golovko et al. (2016)</xref>.</p>
<p><bold>Theorem 3:</bold> Maximization of the log-likelihood input data distribution P(x) in the space of synaptic weights of the restricted Boltzmann machine is equivalent to special case of minimizing the reconstruction mean squared error in the same space.</p>
<p>As a result, the following training rule was obtained for online learning <xref ref-type="disp-formula" rid="EQ15">Eq. (15)</xref>:</p>
<disp-formula id="EQ15">
<label>(15)</label>
<mml:math id="M27">
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="E6">
<mml:math id="M28">
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="E7">
<mml:math id="M29">
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>where &#x03B1; is learning rate.</p>
<p>It is easy to show, that if</p>
<disp-formula id="E8">
<mml:math id="M30">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mtext>,</mml:mtext>
</mml:math>
</disp-formula>
<p>then can be obtained the conventional learning rule <xref ref-type="disp-formula" rid="EQ16">Eq. (16)</xref>:</p>
<disp-formula id="E9">
<mml:math id="M31">
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mtext>,</mml:mtext>
</mml:math>
</disp-formula>
<disp-formula id="EQ16">
<label>(16)</label>
<mml:math id="M32">
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="E10">
<mml:math id="M33">
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mtext>.</mml:mtext>
</mml:math>
</disp-formula>
<p>We have seen in this section, that depending on the loss function can be obtained different learning rules with derivatives and without derivatives of activation function with respect to weighted sum. In further we will use the learning rule with derivatives.</p>
</sec>
<sec id="sec4">
<label>4</label>
<title>Learning of RBM with ReLU</title>
<p>In this section, we consider the definition of ReLU activation function and RBM learning rule. As noted earlier, we consider RBM with deterministic neurons and for learning we will use the expressions given in the previous section. First of all, let us define the ReLU activation function by the following way.</p>
<p>Definition: The ReLU activation function for j-th unit can be presented by the following way <xref ref-type="disp-formula" rid="EQ17">Eq. (17)</xref>:</p>
<disp-formula id="EQ17">
<label>(17)</label>
<mml:math id="M34">
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mi>F</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math id="M35">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:math>
</inline-formula> is defined by the following way <xref ref-type="disp-formula" rid="EQ18">Eq. (18)</xref>:</p>
<disp-formula id="EQ18">
<label>(18)</label>
<mml:math id="M36">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="center">
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>0.</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>Here <inline-formula>
<mml:math id="M37">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M38">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> denote corresponding slopes of the ReLU function; <inline-formula>
<mml:math id="M39">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2260;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>; <inline-formula>
<mml:math id="M40">
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>1.</mml:mn>
</mml:math>
</inline-formula> Usually <inline-formula>
<mml:math id="M41">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> is used.</p>
<p>The above definition of the activation function allows the use of any slope of straight lines and generalizes the conventional definition of ReLU and leaky ReLU activation functions.</p>
<p>Then we can obtain the following derivatives <xref ref-type="disp-formula" rid="EQ19">Eq. (19)</xref>:</p>
<disp-formula id="EQ19">
<label>(19)</label>
<mml:math id="M42">
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mspace width="0.25em"/>
<mml:mi mathvariant="italic">and</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="0.25em"/>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>Using the previous results <xref ref-type="disp-formula" rid="EQ15">Eq. (15)</xref>, we can write the following equations for online learning <xref ref-type="disp-formula" rid="EQ20">Eq. (20)</xref>:</p>
<disp-formula id="EQ20">
<label>(20)</label>
<mml:math id="M43">
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="E11">
<mml:math id="M44">
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="E12">
<mml:math id="M45">
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>If we apply the batch learning and batch size is <inline-formula>
<mml:math id="M46">
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>,</p>
<disp-formula id="EQ21">
<label>(21)</label>
<mml:math id="M47">
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:mfenced open="(" close=")">
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="E13">
<mml:math id="M48">
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="E14">
<mml:math id="M49">
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>Thus, in this section, we have derived learning rules for RBM with ReLU activation function. Further we will use given above expressions <xref ref-type="disp-formula" rid="EQ21">Eq. (21)</xref> for RBM learning.</p>
</sec>
<sec sec-type="materials|methods" id="sec5">
<label>5</label>
<title>Materials and methods</title>
<p>In this section we address adaptive learning rate estimation for RBM with ReLU activation function. Since the RBM network has symmetric weights in the hidden and visible layers, we should derive the optimal training step for the two layers.</p>
<p>The learning step is called adaptive, which is chosen at each stage of the algorithm in such a way in order to minimize the total mean squared error (<xref ref-type="bibr" rid="ref21">Golovko et al., 2000</xref>; <xref ref-type="bibr" rid="ref16">Golovko, 2003</xref>; <xref ref-type="bibr" rid="ref20">Golovko et al., 2023</xref>). We will use the steepest descent approach in order to obtain the expression for adaptive learning rate. Accordingly, to steepest descent approach, the learning step &#x03B1; is selected so as to minimize the mean square error of the new parameters <xref ref-type="disp-formula" rid="EQ22">Eq. (22)</xref>:</p>
<disp-formula id="EQ22">
<label>(22)</label>
<mml:math id="M50">
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:munder accentunder="true">
<mml:mo>min</mml:mo>
<mml:mrow>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math id="M51">
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> are the outputs of the hidden and visible layer at the next time <inline-formula>
<mml:math id="M52">
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> after updating the RBM trainable parameters.</p>
<p>As a result, at each step of learning algorithm we should choose the value of learning rate in such a way that, when modifying weights and thresholds to guarantee a minimum of the mean squared error for each batch or each example from the training data set.</p>
<p><bold>Theorem 4:</bold> For an RBM network with a ReLU activation function in the case of batch learning, the value of the adaptive learning step, that minimizes the mean squared error for each batch is calculated based on the following expression:</p>
<disp-formula id="EQ23">
<label>(23)</label>
<mml:math id="M53">
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:msubsup>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:msubsup>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:msubsup>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:msubsup>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mfrac>
</mml:math>
</disp-formula>
<p>where the corresponding terms are determined according to the expressions <xref ref-type="disp-formula" rid="EQ24 EQ25 EQ26 EQ27 EQ28 EQ29 EQ30 EQ31 EQ32">Eqs. (24&#x2013;32)</xref>:</p>
<disp-formula id="EQ24">
<label>(24)</label>
<mml:math id="M54">
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:math>
</disp-formula>
<disp-formula id="EQ25">
<label>(25)</label>
<mml:math id="M55">
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:math>
</disp-formula>
<disp-formula id="EQ26">
<label>(26)</label>
<mml:math id="M56">
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:math>
</disp-formula>
<disp-formula id="EQ27">
<label>(27)</label>
<mml:math id="M57">
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula><disp-formula id="EQ28">
<label>(28)</label>
<mml:math id="M58">
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:math>
</disp-formula><disp-formula id="EQ29">
<label>(29)</label>
<mml:math id="M59">
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mtext>,</mml:mtext>
</mml:math>
</disp-formula><disp-formula id="EQ30">
<label>(30)</label>
<mml:math id="M60">
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula><disp-formula id="EQ31">
<label>(31)</label>
<mml:math id="M61">
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:math>
</disp-formula><disp-formula id="EQ32">
<label>(32)</label>
<mml:math id="M62">
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="center">
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="center">
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>0.</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p><bold>Proof:</bold> We should find adaptive learning rate by minimizing the following loss function <xref ref-type="disp-formula" rid="EQ33">Eq. (33)</xref>:</p>
<disp-formula id="EQ33">
<label>(33)</label>
<mml:math id="M63">
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="0.25em"/>
<mml:mo>+</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>The output of the hidden and visible layer at the next time <inline-formula>
<mml:math id="M64">
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> after updating trainable parameters can be defined as according to the expressions <xref ref-type="disp-formula" rid="EQ34">Eq. (34)</xref>:</p>
<disp-formula id="EQ34">
<label>(34)</label>
<mml:math id="M65">
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>Let us consider at the beginning the weighted sum of the hidden layer at the next time <inline-formula>
<mml:math id="M66">
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula></p>
<disp-formula id="EQ35">
<label>(35)</label>
<mml:math id="M67">
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>Substituting corresponding expression for weights and threshold updating from <xref ref-type="disp-formula" rid="EQ21">Eq. (21)</xref> in <xref ref-type="disp-formula" rid="EQ35">Eq. (35)</xref> we can obtain <xref ref-type="disp-formula" rid="EQ36">Eq. (36)</xref></p>
<disp-formula id="EQ36">
<label>(36)</label>
<mml:math id="M68">
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:math>
</disp-formula>
<p>where <italic>bj</italic> is defined using <xref ref-type="disp-formula" rid="EQ37">Eq. (37)</xref></p>
<disp-formula id="EQ37">
<label>(37)</label>
<mml:math id="M69">
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="0.25em"/>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>We will use a similar approach for the visible layer. Then the weighted sum of the visible layer can be defined as follows:</p>
<disp-formula id="EQ38">
<label>(38)</label>
<mml:math id="M70">
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>Substituting corresponding expression from <xref ref-type="disp-formula" rid="EQ21">Eq. (21)</xref> in <xref ref-type="disp-formula" rid="EQ38">Eq. (38)</xref> we can write obtain <xref ref-type="disp-formula" rid="EQ39">Eq. (39)</xref></p>
<disp-formula id="EQ39">
<label>(39)</label>
<mml:math id="M71">
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:math>
</disp-formula>
<p>Where <italic>bj</italic> is defined using <xref ref-type="disp-formula" rid="EQ40">Eq. (40)</xref></p>
<disp-formula id="EQ40">
<label>(40)</label>
<mml:math id="M72">
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="0.25em"/>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>As a result, we can obtain the final expressions regarding output of the hidden and visible layer <xref ref-type="disp-formula" rid="EQ41">Eq. (41)</xref></p>
<disp-formula id="EQ41">
<label>(41)</label>
<mml:math id="M73">
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="E16">
<mml:math id="M74">
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mtext>.</mml:mtext>
</mml:math>
</disp-formula>
<p>Differentiating the loss function <inline-formula>
<mml:math id="M75">
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> with respect to &#x03B1; we can obtain <xref ref-type="disp-formula" rid="EQ42">Eq. (42)</xref></p>
<disp-formula id="EQ42">
<label>(42)</label>
<mml:math id="M76">
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mfrac>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>&#x03B1;</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="2.75em"/>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:munderover>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mfenced open="(" close=")">
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="2.75em"/>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
<mml:mspace width="thickmathspace"/>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>As a result, we can obtain the following final expression <xref ref-type="disp-formula" rid="EQ43">Eq. (43)</xref>:</p>
<disp-formula id="EQ43">
<label>(43)</label>
<mml:math id="M77">
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:msubsup>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:msubsup>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:msubsup>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:msubsup>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mfrac>
</mml:math>
</disp-formula>
<p>Since in accordance with <xref ref-type="disp-formula" rid="EQ44">Eq. (44)</xref></p>
<disp-formula id="EQ44">
<label>(44)</label>
<mml:math id="M78">
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>d</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mi>&#x03B1;</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x003E;</mml:mo>
<mml:mn>0</mml:mn>
</mml:math>
</disp-formula>
<p>we have found the minimum of the cost function. Thus the theorem is proved.</p>
<p>As follows from the proven theorem, the adaptive learning rate minimizes the mean squared error of the network under updating weights and thresholds.</p>
<p>The major difficulty arises in the computing of <inline-formula>
<mml:math id="M79">
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, because it is desired parameters of ReLU transfer function. Since the desired outputs of the hidden and visible layer correspondingly <inline-formula>
<mml:math id="M80">
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M81">
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:math>
</inline-formula> then we can write <xref ref-type="disp-formula" rid="EQ45">Eq. (45)</xref></p>
<disp-formula id="EQ45">
<label>(45)</label>
<mml:math id="M82">
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="center">
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="center">
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>0.</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p><bold>Theorem 5:</bold> For RBM network with a ReLU activation function in the case of online learning, the value of the adaptive learning step, that minimizes the mean squared error for each pattern is defined as follows <xref ref-type="disp-formula" rid="EQ46">Eq. (46)</xref>:</p>
<disp-formula id="EQ46">
<label>(46)</label>
<mml:math id="M83">
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<p>where the corresponding terms are calculated based on <xref ref-type="disp-formula" rid="EQ47">Eqs. (47</xref>&#x2013;<xref ref-type="disp-formula" rid="EQ55">55)</xref></p>
<disp-formula id="EQ47">
<label>(47)</label>
<mml:math id="M84">
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<disp-formula id="EQ48">
<label>(48)</label>
<mml:math id="M85">
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<disp-formula id="EQ49">
<label>(49)</label>
<mml:math id="M86">
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<disp-formula id="EQ50">
<label>(50)</label>
<mml:math id="M87">
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ51">
<label>(51)</label>
<mml:math id="M88">
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ52">
<label>(52)</label>
<mml:math id="M89">
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<disp-formula id="EQ53">
<label>(53)</label>
<mml:math id="M90">
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ54">
<label>(54)</label>
<mml:math id="M91">
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:math>
</disp-formula><disp-formula id="EQ55">
<label>(55)</label>
<mml:math id="M92">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="center">
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="center">
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="center">
<mml:mtd columnalign="center">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>0.</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>This theorem is proved by the same approach.</p>
<p>It should be noted that the proposed expressions for calculating the learning step are valid when <inline-formula>
<mml:math id="M93">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2260;</mml:mo>
<mml:mn>0.</mml:mn>
</mml:math>
</inline-formula> If <inline-formula>
<mml:math id="M94">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:math>
</inline-formula>, then in accordance with RBM learning rule <xref ref-type="disp-formula" rid="EQ21">Eq. (21)</xref>, the training is performed only in the area where weighted sum is greater than 0, since the gradient of this function is 0, if weighted sum less than 0. In that case, we can simplify the expressions for learning rate.</p>
<p><bold>Corollary:</bold> For RBM network with a ReLU activation function and <inline-formula>
<mml:math id="M95">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:math>
</inline-formula> in the case of online learning, the value of the adaptive learning step is calculated based on the following expression <xref ref-type="disp-formula" rid="EQ56 EQ57 EQ58">Eqs. (56&#x2013;58)</xref>:</p>
<disp-formula id="EQ56">
<label>(56)</label>
<mml:math id="M96">
<mml:mi>&#x03B1;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<p>Where</p>
<disp-formula id="EQ57">
<label>(57)</label>
<mml:math id="M97">
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo stretchy="true">(</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ58">
<label>(58)</label>
<mml:math id="M98">
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo stretchy="true">(</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>0</mml:mn>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>In a similar way, we can obtain an expression for the adaptive step calculation when using batch learning. The proposed expressions allow to estimate the learning rate after presenting every batch or pattern to the neural network and based on the minimization of loss function. Adaptive training step approach permits to choose automatically step for every batch or pattern from training data set. The performance of proposed approach is discussed in the next section.</p>
</sec>
<sec id="sec6">
<label>6</label>
<title>Experiments</title>
<p>This section summarizes numerical results obtained by the application of adaptive and constant learning rate. In order to evaluate the performance of the proposed approach we will conduct various experiments using RBM and deep neural network. In all experiments, we will use batch learning with adaptive rate <xref ref-type="disp-formula" rid="EQ23">Eq. (23)</xref>. In that case the weights and thresholds of the network will be modified based on rule <xref ref-type="disp-formula" rid="EQ21">Eq. (21)</xref> presented in this paper. For experiments, we will use both an artificial and the MNIST dataset. The primary aim of this section is to compare learning of neural network with and without proposed training approach with adaptive learning rate. The experiments are divided into 2 groups. The first experiments focuses on the RBM network and the second on deep multilayer neural network.</p>
<sec id="sec7">
<label>6.1</label>
<title>RBM results</title>
<p>Let us consider the use of an adaptive learning step for a RBM network. To evaluate the effectiveness of adaptive learning rate we will use two datasets.</p>
<sec id="sec8">
<label>6.1.1</label>
<title>Artificial dataset</title>
<p>The artificial data x lie on a one-dimensional manifold (a helical loop) embedded in three dimensions (<xref ref-type="bibr" rid="ref42">Scholz et al., 2008</xref>) and were generated from a uniformly distributed factor t in the range [0.05, 0.95]:</p>
<disp-formula id="EQ59">
<mml:math id="M104">
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>sin</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x03C0;</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>&#x03BC;</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>cos</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x03C0;</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>&#x03BC;</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>&#x03BC;</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mspace width="0.25em"/>
<mml:mtext>,</mml:mtext>
</mml:math>
</disp-formula>
<p>where &#x03BC; &#x2013; Gaussian noise with mean 0 and standard deviation 0.05.</p>
<p>The primary goal of the experiment is to study the performance of ATS for data compression and reconstruction. Then the RBM will consist of 3 visible and 1 hidden unit. The training dataset consists of 1,000 samples. The size of the test patterns is also 1,000. The batch size equal 8 and the parameters of ReLU function are the following: <inline-formula>
<mml:math id="M99">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>0.01</mml:mn>
<mml:mtext>.</mml:mtext>
</mml:math>
</inline-formula> We trained the RBM network using only clean data and tested using noisy data. The evolution of reconstruction error vs. epoch of RBM learning is provided in <xref ref-type="table" rid="tab1">Table 1</xref>. As can be seen from the table the adaptive learning rate has obvious excellence compared to constant steps. The plots of the reconstruction accuracy vs. epoch for learning and testing using the best constant and adaptive rate are presented in <xref ref-type="fig" rid="fig4">Figures 4</xref>, <xref ref-type="fig" rid="fig5">5</xref>. It should be noted here that testing is performed after each learning epoch. As follows from the presented figures, the adaptive learning rate has the evident advantage compared to the fixed learning rate, namely, the best performance in terms of learning quality and generalization ability.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Evolution of the reconstruction error (MSE) for artificial data.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Number of epochs</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 1</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 2</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 3</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 4</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 5</th>
<th align="center" valign="top">ATS</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">10</td>
<td align="center" valign="top">0.2223</td>
<td align="center" valign="top">0.0974</td>
<td align="center" valign="top">0.1394</td>
<td align="center" valign="top">0.1393</td>
<td align="center" valign="top">0.1862</td>
<td align="center" valign="top">0.1001</td>
</tr>
<tr>
<td align="left" valign="top">20</td>
<td align="center" valign="top">0.2240</td>
<td align="center" valign="top">0.0975</td>
<td align="center" valign="top">0.0982</td>
<td align="center" valign="top">0.1392</td>
<td align="center" valign="top">0.1653</td>
<td align="center" valign="top">0.0903</td>
</tr>
<tr>
<td align="left" valign="top">30</td>
<td align="center" valign="top">0.2241</td>
<td align="center" valign="top">0.0982</td>
<td align="center" valign="top">0.0983</td>
<td align="center" valign="top">0.1392</td>
<td align="center" valign="top">0.1537</td>
<td align="center" valign="top">0.0902</td>
</tr>
<tr>
<td align="left" valign="top">40</td>
<td align="center" valign="top">0.2230</td>
<td align="center" valign="top">0.0972</td>
<td align="center" valign="top">0.0982</td>
<td align="center" valign="top">0.1392</td>
<td align="center" valign="top">0.1474</td>
<td align="center" valign="top">0.0737</td>
</tr>
<tr>
<td align="left" valign="top">50</td>
<td align="center" valign="top">0.2231</td>
<td align="center" valign="top">0.0972</td>
<td align="center" valign="top">0.0982</td>
<td align="center" valign="top">0.1391</td>
<td align="center" valign="top">0.1422</td>
<td align="center" valign="top">0.0753</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Plot of reconstruction accuracy versus epoch for learning using adaptive and constant learning rates.</p>
</caption>
<graphic xlink:href="fnins-18-1362510-g004.tif"/>
</fig>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Plot of reconstruction accuracy versus epoch for testing using adaptive and constant learning rates.</p>
</caption>
<graphic xlink:href="fnins-18-1362510-g005.tif"/>
</fig>
</sec>
<sec id="sec9">
<label>6.1.2</label>
<title>MNIST dataset</title>
<p>In this section we will use the MNIST dataset, which contains 60,000 hand-written digit images for training, and 10,000 images for testing. Data in MNIST are grayscale images with size 28&#x2009;&#x00D7;&#x2009;28. Before training the images are normalized to be zero-mean.</p>
<p>Let us model the RBM network. This simulation is used to illustrate the compression and reconstruction properties of restricted Boltzmann machine. Let us model the RBM network which consist of 784 neurons of visible and 128&#x2009;units of hidden layers. The main goal of such modeling is to compress and reconstruct MNIST data. The parameters of experiments are shown in <xref ref-type="table" rid="tab2">Table 2</xref>. We used original images from MNIST dataset and before representation to RBM only centering is performed.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Parameters of experiments.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="middle">Number of pretraining epoch</th>
<th align="center" valign="middle">Batch size</th>
<th align="center" valign="middle">
<inline-formula>
<mml:math id="M100">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>
</th>
<th align="center" valign="middle">
<inline-formula>
<mml:math id="M101">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>
</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">8</td>
<td align="center" valign="top">128</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">0.01</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The evolution of reconstruction square error <xref ref-type="disp-formula" rid="EQ13">Eq. (13)</xref> is shown in <xref ref-type="table" rid="tab3">Table 3</xref>. Here div. Denotes divergence of learning. The analysis of the data in this table indicates that learning with a constant rate is unstable. For instance, if &#x03B1;&#x2009;=&#x2009;1e &#x2212; 4, the neural network cannot be trained. As can be seen only training with a constant learning rate (3e &#x2212; 7) leads to a positive result.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Evolution of the reconstruction error (MSE) for MNIST.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Number of epochs</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;1e &#x2212; 4</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 5</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 6</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 7</th>
<th align="center" valign="top">ATS</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">1</td>
<td align="center" valign="top">0,126</td>
<td align="center" valign="top">0.095</td>
<td align="center" valign="top">0.073</td>
<td align="center" valign="top">0.073787208</td>
<td align="center" valign="top">0.070819163</td>
</tr>
<tr>
<td align="left" valign="top">2</td>
<td align="center" valign="top">0,178</td>
<td align="center" valign="top">0.111</td>
<td align="center" valign="top">0.074</td>
<td align="center" valign="top">0.073663836</td>
<td align="center" valign="top">0.070752306</td>
</tr>
<tr>
<td align="left" valign="top">3</td>
<td align="center" valign="top">0,213</td>
<td align="center" valign="top">0.125</td>
<td align="center" valign="top">0.075</td>
<td align="center" valign="top">0.073577446</td>
<td align="center" valign="top">0.070733909</td>
</tr>
<tr>
<td align="left" valign="top">4</td>
<td align="center" valign="top">div.</td>
<td align="center" valign="top">0.138</td>
<td align="center" valign="top">0.077</td>
<td align="center" valign="top">0.073518835</td>
<td align="center" valign="top">0.070713023</td>
</tr>
<tr>
<td align="left" valign="top">5</td>
<td align="center" valign="top">div.</td>
<td align="center" valign="top">0.154</td>
<td align="center" valign="top">0.081</td>
<td align="center" valign="top">0.073481718</td>
<td align="center" valign="top">0.070700173</td>
</tr>
<tr>
<td align="left" valign="top">6</td>
<td align="center" valign="top">div.</td>
<td align="center" valign="top">0.171</td>
<td align="center" valign="top">0.087</td>
<td align="center" valign="top">0.073461681</td>
<td align="center" valign="top">0.070700861</td>
</tr>
<tr>
<td align="left" valign="top">7</td>
<td align="center" valign="top">div.</td>
<td align="center" valign="top">0.199</td>
<td align="center" valign="top">0.098</td>
<td align="center" valign="top">0.073455503</td>
<td align="center" valign="top">0.070687373</td>
</tr>
<tr>
<td align="left" valign="top">8</td>
<td align="center" valign="top">div.</td>
<td align="center" valign="top">0.232</td>
<td align="center" valign="top">0.115</td>
<td align="center" valign="top">0.073460822</td>
<td align="center" valign="top">0.070678658</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Hence, the learning algorithms with constant training step can diverge if the learning parameters are not chosen appropriately, as shown in <xref ref-type="table" rid="tab3">Table 3</xref>. Therefore, we should select the constant learning rate very carefully.</p>
<p>Also it should be remarked, that learning with ATS have shown the result after first epoch better than with constant step at any epoch. After 8 epochs have obtained the best result with reconstruction error of 0.070678658. The best result using constant step is 0.073455503. This result was obtained after 8 epochs. As can be seen, the adaptive learning rate has a significant advantage in comparison with the constant learning stage. The evolution of reconstruction error is presented in <xref ref-type="fig" rid="fig6">Figure 6</xref>.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Evolution of reconstruction error MSE for adaptive and constant learning rates.</p>
</caption>
<graphic xlink:href="fnins-18-1362510-g006.tif"/>
</fig>
</sec>
</sec>
<sec id="sec10">
<label>6.2</label>
<title>Deep multilayer perceptron</title>
<p>Let us consider the analysis of the proposed approach for a deep multilayer neural network using the MNIST dataset. We have used for experiments deep perceptron with ReLU activation function which has the following structure: 784-1600-1600-800-800-10. The parameters of experiments are shown in <xref ref-type="table" rid="tab4">Table 4</xref>. The results of our experiments are shown in <xref ref-type="table" rid="tab5">Table 5</xref>. Pretraining is performed using only 1&#x2013;3 epochs.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>MNIST Classification Experiment Parameters.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="middle">Number of pretraining epochs</th>
<th align="center" valign="middle">Number of fine-tuning epochs</th>
<th align="center" valign="middle">Batch size</th>
<th align="center" valign="middle">
<inline-formula>
<mml:math id="M102">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>
</th>
<th align="center" valign="middle">
<inline-formula>
<mml:math id="M103">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>
</th>
<th align="center" valign="middle">Initial &#x03B1; at the finetuning stage</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">3</td>
<td align="center" valign="top">50</td>
<td align="center" valign="top">128</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">0.01</td>
<td align="center" valign="top">3e &#x2212; 4</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption>
<p>Testing a deep multilayer perceptron pertaining.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;1e &#x2212; 4</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 5</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 6</th>
<th align="center" valign="top">&#x03B1;&#x2009;=&#x2009;3e &#x2212; 7</th>
<th align="center" valign="top">ADAM</th>
<th align="center" valign="top">ATS</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">MSE</td>
<td align="center" valign="top">div</td>
<td align="center" valign="top">div</td>
<td align="center" valign="top">0.0024235</td>
<td align="center" valign="top">0.0024619</td>
<td align="center" valign="top">0.0024729</td>
<td align="center" valign="top">0.0022999</td>
</tr>
<tr>
<td align="left" valign="top">Precision macro</td>
<td align="center" valign="top">0.0098</td>
<td align="center" valign="top">0.0098</td>
<td align="center" valign="top">0.986</td>
<td align="center" valign="top">0.98562</td>
<td align="center" valign="top">0.98539</td>
<td align="center" valign="top">0.98656</td>
</tr>
<tr>
<td align="left" valign="top">Accuracy</td>
<td align="center" valign="top">0.0098</td>
<td align="center" valign="top">0.0098</td>
<td align="center" valign="top">0.986</td>
<td align="center" valign="top">0.9857</td>
<td align="center" valign="top">0.9854</td>
<td align="center" valign="top">0.9866</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The evolution of mean squared error for different approaches is presented in <xref ref-type="fig" rid="fig7">Figure 7</xref>. Finally, we have the following experimental results, which are shown in <xref ref-type="table" rid="tab5">Table 5</xref>. The smallest test error without using ATS and pretraining is 0.002531. If we use ATS the smallest test error is 0.002299. The smallest test error for pretraining with constant step is 0.002392. As in the previous case, the table shows that learning with a constant rate can be unstable. As a result, a learning algorithm with a constant learning step may be divergent. <xref ref-type="table" rid="tab6">Tables 6</xref>&#x2013;<xref ref-type="table" rid="tab8">8</xref> show the predictive performance of different learning approaches for MNIST classification. As can be seen, in general the ATS approach outperformed the learning technique with fixed learning rate. So, for instance, the number of correct predictions using the adaptive rate (<xref ref-type="table" rid="tab6">Table 6</xref>) is 6 and 9 more, respectively, compared to models with a constant step (<xref ref-type="table" rid="tab7">Tables 7</xref>, <xref ref-type="table" rid="tab8">8</xref>). As can be seen the use of ATS permits to reduce the test error and correspondingly improve the generalization ability.</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Evolution of MSE for adaptive and constant learning rates.</p>
</caption>
<graphic xlink:href="fnins-18-1362510-g007.tif"/>
</fig>
<table-wrap position="float" id="tab6">
<label>Table 6</label>
<caption>
<p>Confusion matrix for MNIST classification using pretraining with ATS.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th colspan="2" rowspan="2"></th>
<th align="center" valign="top" colspan="10">Predicted values</th>
</tr>
<tr>
<th align="center" valign="top">0</th>
<th align="center" valign="top">1</th>
<th align="center" valign="top">2</th>
<th align="center" valign="top">3</th>
<th align="center" valign="top">4</th>
<th align="center" valign="top">5</th>
<th align="center" valign="top">6</th>
<th align="center" valign="top">7</th>
<th align="center" valign="top">8</th>
<th align="center" valign="top">9</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="10">Actual values</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">974</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1,130</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1,015</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">5</td>
<td align="center" valign="middle">5</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">998</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">966</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">10</td>
</tr>
<tr>
<td align="center" valign="middle">5</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">5</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">877</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">947</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">7</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1,013</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">4</td>
</tr>
<tr>
<td align="center" valign="middle">8</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">5</td>
<td align="center" valign="middle">960</td>
<td align="center" valign="middle">3</td>
</tr>
<tr>
<td align="center" valign="middle">9</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">7</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">986</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab7">
<label>Table 7</label>
<caption>
<p>Confusion matrix for MNIST classification using pretraining with &#x03B1;&#x2009;=&#x2009;3e &#x2212; 6.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th colspan="2" rowspan="2"></th>
<th align="center" valign="top" colspan="10">Predicted values</th>
</tr>
<tr>
<th align="center" valign="top">0</th>
<th align="center" valign="top">1</th>
<th align="center" valign="top">2</th>
<th align="center" valign="top">3</th>
<th align="center" valign="top">4</th>
<th align="center" valign="top">5</th>
<th align="center" valign="top">6</th>
<th align="center" valign="top">7</th>
<th align="center" valign="top">8</th>
<th align="center" valign="top">9</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="10">Actual values</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">973</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1,126</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1,018</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">7</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">999</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">2</td>
</tr>
<tr>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">969</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">7</td>
</tr>
<tr>
<td align="center" valign="middle">5</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">5</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">878</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">2</td>
</tr>
<tr>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">5</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">945</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">7</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">7</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1,010</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">3</td>
</tr>
<tr>
<td align="center" valign="middle">8</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">7</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">950</td>
<td align="center" valign="middle">2</td>
</tr>
<tr>
<td align="center" valign="middle">9</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">992</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab8">
<label>Table 8</label>
<caption>
<p>Confusion matrix for MNIST classification using pretraining with &#x03B1;&#x2009;=&#x2009;3e-7.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th colspan="2" rowspan="2"></th>
<th align="center" valign="top" colspan="10">Predicted values</th>
</tr>
<tr>
<th align="center" valign="top">0</th>
<th align="center" valign="top">1</th>
<th align="center" valign="top">2</th>
<th align="center" valign="top">3</th>
<th align="center" valign="top">4</th>
<th align="center" valign="top">5</th>
<th align="center" valign="top">6</th>
<th align="center" valign="top">7</th>
<th align="center" valign="top">8</th>
<th align="center" valign="top">9</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="10">Actual values</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">974</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1,126</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1,021</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">999</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">5</td>
</tr>
<tr>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">964</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">7</td>
</tr>
<tr>
<td align="center" valign="middle">5</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">881</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">940</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">7</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">5</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">1,015</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
</tr>
<tr>
<td align="center" valign="middle">8</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">952</td>
<td align="center" valign="middle">4</td>
</tr>
<tr>
<td align="center" valign="middle">9</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">4</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">2</td>
<td align="center" valign="middle">985</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="sec11">
<label>7</label>
<title>Conclusion and discussion</title>
<p>The learning of neural networks is a tricky task, which highly depends on suitable hyperparameters selection to achieve significant performance of a neural network. The choice of an appropriate learning rate is of great importance because it has a significant impact on the training efficiency. Depending on this parameter the learning process can be divergent or convergent.</p>
<p>We have not found any works as concerns exact analytical expressions for the learning rate estimation, based on the steepest descent technique. Such precise analytical expressions can only be obtained for linear and ReLU activation functions. When using the sigmoid activation function, we can only receive approximate expressions for the learning rate using the Taylor series expansion. Since this is a very complicated problem, most of the scientists use the steepest descent method together with the line search approach.</p>
<p>Our previous work (<xref ref-type="bibr" rid="ref20">Golovko et al., 2023</xref>) reported an adaptive learning rate for a single-layer perceptron with a ReLU activation function. In this work, we extended this idea to obtain the learning rate for the RBM network. As a result, novel analytical expressions for learning step estimation have been proposed in this paper. The proposed approach for ATS estimation is based on minimization the mean squared error for each batch or each sample. The presented expressions are applied for restricted Boltzmann machine learning with ReLU activation function. We consider quasi-conventional RBM, namely we use symmetric weights in the hidden and visible layers, Gibbs sampling and deterministic units. We first demonstrate the proposed approach for ATS calculation is more effective and more efficient for RBM learning than the conventional RBM algorithm. Second, we show that such kind of RBM can be used for deep neural network pretraining using greedy layer wise algorithm. As a result, we can reach better generalization ability.</p>
<p>The main advantages of the proposed approach are the following: it is based on precise mathematical expressions obtained by minimizing the mean squared error for each batch or each pattern; it is capable of automatically defining and adjusting the learning rate during neural network training; it guarantees convergence to well-performing local minima. The disadvantage of the presented approach is higher computational complexity compared to constant step.</p>
<p>This work opens the way toward the following future research: define the conditions where such an learning rate can guarantee convergence to best-performing local minima and to study how this approach can be extended to train a multilayer neural network without pretraining using RBM.</p>
</sec>
<sec sec-type="data-availability" id="sec12">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found at: <ext-link xlink:href="https://yann.lecun.com/exdb/mnist/" ext-link-type="uri">https://yann.lecun.com/exdb/mnist/</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="sec13">
<title>Author contributions</title>
<p>CC: Data curation, Funding acquisition, Investigation, Writing &#x2013; original draft. VG: Conceptualization, Formal analysis, Supervision, Writing &#x2013; original draft. AK: Methodology, Project administration, Visualization, Writing &#x2013; review &#x0026; editing. EM: Software, Validation, Writing &#x2013; review &#x0026; editing. MC: Funding acquisition, Investigation, Writing &#x2013; review &#x0026; editing. PL: Methodology, Resources, Software, Writing &#x2013; review &#x0026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec14">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This research was partially supported by the Ministry of Science and Technology of the People&#x2019;s Republic of China (grant number G2022016010L) and Belarusian Republican Foundation for Fundamental Research (grant &#x0424;22&#x041A;&#x0418;-046).</p>
</sec>
<sec sec-type="COI-statement" id="sec15">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec100" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aguilera</surname> <given-names>A.</given-names></name> <name><surname>Olmos</surname> <given-names>P.</given-names></name> <name><surname>Art&#x00E9;s-Rodr&#x00ED;guez</surname> <given-names>A.</given-names></name> <name><surname>P&#x00E9;rez-Cruz</surname> <given-names>F.</given-names></name></person-group> (<year>2023</year>). <article-title>Regularizing transformers with deep probabilistic layers</article-title>. <source>Neural Netw.</source> <volume>161</volume>, <fpage>565</fpage>&#x2013;<lpage>574</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neunet.2023.01.032</pub-id>, PMID: <pub-id pub-id-type="pmid">36812832</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Arpit</surname> <given-names>D.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). The benefits of overparameterization at initialization in deep ReLU networks. arXiv [Preprint]. arXiv:1901.03611.</citation></ref>
<ref id="ref3"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Baydin</surname> <given-names>A. G.</given-names></name> <name><surname>Cornish</surname> <given-names>R.</given-names></name> <name><surname>Rubio</surname> <given-names>D. M.</given-names></name> <name><surname>Schmidt</surname> <given-names>M.</given-names></name> <name><surname>Wood</surname> <given-names>F.</given-names></name></person-group> (<year>2018</year>). Online learning rate adaptation with hypergradient descent. In <italic>Proceedings of International Conference on Learning Representations</italic>.</citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>2009</year>). <article-title>Learning deep architectures for AI</article-title>. <source>Foundat Trends Machine Learn</source> <volume>2</volume>, <fpage>1</fpage>&#x2013;<lpage>127</lpage>. doi: <pub-id pub-id-type="doi">10.1561/2200000006</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Courville</surname> <given-names>A.</given-names></name> <name><surname>Vincent</surname> <given-names>P.</given-names></name></person-group> (<year>2013</year>). Representation learning a review and new perspectives. In: <italic>Institute of Electrical and Electronics Engineers transactions on pattern analysis and machine intelligence</italic>; 35, 1798&#x2013;1828.</citation></ref>
<ref id="ref6"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Lamblin</surname> <given-names>P.</given-names></name> <name><surname>Popovici</surname> <given-names>D.</given-names></name> <name><surname>Larochelle</surname> <given-names>H.</given-names></name></person-group> (<year>2007</year>). &#x201C;<article-title>Greedy layer-wise training of deep networks</article-title>&#x201D; in <source>Advances in neural information processing systems</source>. eds. <person-group person-group-type="editor"><name><surname>Scholkopf</surname> <given-names>J. C.</given-names></name> <name><surname>Platt</surname> <given-names>T.</given-names></name> <name><surname>Hoffman</surname> <given-names>S.</given-names></name></person-group>, MIT Press, Cambridge vol. <volume>11</volume>, <fpage>153</fpage>&#x2013;<lpage>160</lpage>.</citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Lecun</surname> <given-names>Y.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep learning for AI</article-title>. <source>Commun. ACM</source> <volume>64</volume>, <fpage>58</fpage>&#x2013;<lpage>65</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3448250</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Carvalho</surname> <given-names>P.</given-names></name> <name><surname>Lourencco</surname> <given-names>N.</given-names></name> <name><surname>Machado</surname> <given-names>P.</given-names></name></person-group> (<year>2021</year>). Evolving learning rate optimizers for deep neural networks. arXiv [Preprint]. arXiv:2103.12623.</citation></ref>
<ref id="ref9"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>B.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Ba</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). Differentiable self-adaptive learning rate. ArXiv [Preprint]. arXiv:2210.10290.</citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>K.</given-names></name> <name><surname>Weng</surname> <given-names>Y.</given-names></name> <name><surname>Hosseini</surname> <given-names>A.</given-names></name> <name><surname>Dening</surname> <given-names>T.</given-names></name> <name><surname>Zuo</surname> <given-names>G.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2024</year>). <article-title>A comparative study of GNN and MLP based machine learning for the diagnosis of Alzheimer&#x2019;s disease involving data synthesis</article-title>. <source>Neural Netw.</source> <volume>169</volume>, <fpage>442</fpage>&#x2013;<lpage>452</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neunet.2023.10.040</pub-id>, PMID: <pub-id pub-id-type="pmid">37939533</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Cho</surname> <given-names>K.</given-names></name> <name><surname>Raiko</surname> <given-names>T.</given-names></name> <name><surname>Ilin</surname> <given-names>A.</given-names></name></person-group> (<year>2011</year>). Enhanced gradient and adaptive learning rate for training restricted Boltzmann machines. In: <italic>Proceedings of the 28th International Conference on Machine Learning</italic>, <italic>ICML</italic>, Bellevue, Washington, USA.</citation></ref>
<ref id="ref12"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Defazio</surname> <given-names>A.</given-names></name> <name><surname>Cutkosky</surname> <given-names>A.</given-names></name> <name><surname>Mehta</surname> <given-names>H.</given-names></name> <name><surname>Mishchenko</surname> <given-names>K.</given-names></name></person-group> (<year>2023</year>). When, why and how much? Adaptive learning rate scheduling by refinement. arXiv [Preprint]. arXiv:2310.07831.</citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duchi</surname> <given-names>J.</given-names></name> <name><surname>Hazan</surname> <given-names>E.</given-names></name> <name><surname>Singer</surname> <given-names>Y.</given-names></name></person-group> (<year>2011</year>). <article-title>Adaptive subgradient methods for online learning and stochastic optimization</article-title>. <source>J. Mach. Learn. Res.</source> <volume>12</volume>, <fpage>257</fpage>&#x2013;<lpage>269</lpage>. doi: <pub-id pub-id-type="doi">10.5555/1953048.2021068</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Erhan</surname> <given-names>D.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Courville</surname> <given-names>A.</given-names></name> <name><surname>Manzagol</surname> <given-names>P.-A.</given-names></name> <name><surname>Vincent</surname> <given-names>P.</given-names></name> <name><surname>Bengio</surname> <given-names>S.</given-names></name></person-group> (<year>2010</year>). <article-title>Why does unsupervised pre-training help deep learning?</article-title> <source>J. Mach. Learn. Res.</source> <volume>11</volume>, <fpage>625</fpage>&#x2013;<lpage>660</lpage>.</citation></ref>
<ref id="ref15"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Glorot</surname> <given-names>X.</given-names></name> <name><surname>Bordes</surname> <given-names>A.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>2011</year>). Deep sparse rectifier networks. In <italic>Proceedings of the 14th international conference on artificial intelligence and statistics</italic>. JMLR W&#x0026;CP; 15, 315&#x2013;323.</citation></ref>
<ref id="ref16"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Golovko</surname> <given-names>V.</given-names></name></person-group> (<year>2003</year>). &#x201C;<article-title>From neural networks to intelligent systems: selected aspects of training, application and evolution</article-title>&#x201D; in ed. Marco Gori. <source>Limitations and future trends in neural computation</source> (<publisher-loc>Amsterdam</publisher-loc>: <publisher-name>IOS Press</publisher-name>), <fpage>219</fpage>&#x2013;<lpage>243</lpage>.</citation></ref>
<ref id="ref17"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Golovko</surname> <given-names>V.</given-names></name> <name><surname>Komar</surname> <given-names>M.</given-names></name> <name><surname>Sachenko</surname> <given-names>A.</given-names></name></person-group> (<year>2010</year>). Principles of neural network artificial immune system design to detect attacks on computers. In <italic>Proceedings of the international Conference on Modern Problems of Radio Engineering (TSET)</italic>, p.<fpage>237</fpage>.</citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Golovko</surname> <given-names>V.</given-names></name> <name><surname>Kroshchanka</surname> <given-names>A.</given-names></name> <name><surname>Treadwell</surname> <given-names>D.</given-names></name></person-group> (<year>2016</year>). <article-title>The nature of unsupervised learning in deep neural networks: a new understanding and novel approach</article-title>. <source>Optic Memory Neural Netw</source> <volume>25</volume>, <fpage>127</fpage>&#x2013;<lpage>141</lpage>. doi: <pub-id pub-id-type="doi">10.3103/S1060992X16030073</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Golovko</surname> <given-names>V.</given-names></name> <name><surname>Kroshchanka</surname> <given-names>A.</given-names></name> <name><surname>Turchenko</surname> <given-names>V.</given-names></name> <name><surname>Jankowski</surname> <given-names>S.</given-names></name> <name><surname>Treadwell</surname> <given-names>D.</given-names></name></person-group> (<year>2015</year>). A new technique for restricted Boltzmann machine learning. In <italic>Proceedings of the 8th IEEE international conference IDAACS</italic>, pp.182&#x2013;186.</citation></ref>
<ref id="ref20"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Golovko</surname> <given-names>V.</given-names></name> <name><surname>Mikhno</surname> <given-names>E.</given-names></name> <name><surname>Kroschanka</surname> <given-names>A.</given-names></name> <name><surname>Chodyka</surname> <given-names>M.</given-names></name> <name><surname>Lichograj</surname> <given-names>P.</given-names></name></person-group> (<year>2023</year>). Adaptive learning rate for unsupervised learning of deep neural networks. <italic>International Joint Conference on Neural Networks (IJCNN)</italic>, pp. 1&#x2013;6.</citation></ref>
<ref id="ref21"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Golovko</surname> <given-names>V.</given-names></name> <name><surname>Savitsky</surname> <given-names>Y.</given-names></name> <name><surname>Laopoulos</surname> <given-names>T.</given-names></name> <name><surname>Sachenko</surname> <given-names>A.</given-names></name> <name><surname>Grandinetti</surname> <given-names>L.</given-names></name></person-group> (<year>2000</year>). Technique of learning rate estimation for efficient training of MLP. In <italic>Proceedings of the IEEE-INNS-ENNS international joint conference on neural networks, IJCNN</italic>. Neural computing: New challenges and perspectives for the new millennium; pp. 323&#x2013;328.</citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2002</year>). <article-title>Training products of experts by minimizing contrastive divergence</article-title>. <source>Neural Comput.</source> <volume>14</volume>, <fpage>1771</fpage>&#x2013;<lpage>1800</lpage>. doi: <pub-id pub-id-type="doi">10.1162/089976602760128018</pub-id>, PMID: <pub-id pub-id-type="pmid">12180402</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2010</year>). <source>A practical guide to training restricted Boltzmann machines</source>, <publisher-loc>Toronto</publisher-loc>: <publisher-name>Machine Learning Group, University of Toronto</publisher-name>.</citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G.</given-names></name> <name><surname>Deng</surname> <given-names>L.</given-names></name> <name><surname>Yu</surname> <given-names>D.</given-names></name> <name><surname>Dahl</surname> <given-names>G.</given-names></name> <name><surname>Mohamed</surname> <given-names>A. R.</given-names></name> <name><surname>Jaitly</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Deep neural networks for acoustic Modeling in speech recognition: the shared views of four research groups</article-title>. <source>IEEE Signal Process. Mag.</source> <volume>29</volume>, <fpage>82</fpage>&#x2013;<lpage>97</lpage>. doi: <pub-id pub-id-type="doi">10.1109/MSP.2012.2205597</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G.</given-names></name> <name><surname>Osindero</surname> <given-names>S.</given-names></name> <name><surname>Teh</surname> <given-names>Y.</given-names></name></person-group> (<year>2006</year>). <article-title>A fast learning algorithm for deep belief nets</article-title>. <source>Neural Comput.</source> <volume>18</volume>, <fpage>1527</fpage>&#x2013;<lpage>1554</lpage>. doi: <pub-id pub-id-type="doi">10.1162/neco.2006.18.7.1527</pub-id>, PMID: <pub-id pub-id-type="pmid">16764513</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G.</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R.</given-names></name></person-group> (<year>2006</year>). <article-title>Reducing the dimensionality of data with neural networks</article-title>. <source>Science</source> <volume>313</volume>, <fpage>504</fpage>&#x2013;<lpage>507</lpage>. doi: <pub-id pub-id-type="doi">10.1126/science.1127647</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D. P.</given-names></name> <name><surname>Ba</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). <source>Adam: A method for stochastic optimization, Computer Science,</source> arXiv preprint.</citation></ref>
<ref id="ref28"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>I. S.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2012</year>). Imagenet classification with deep convolutional neural networks. In <italic>Advances in neural information processing systems</italic>, 1097&#x2013;1105.</citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>L.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2012</year>). <article-title>Image net classification with deep convolutional neural networks. In proc</article-title>. <source>Adv. Neural Inf. Proces. Syst.</source> <volume>25</volume>, <fpage>1090</fpage>&#x2013;<lpage>1098</lpage>.</citation></ref>
<ref id="ref30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lamb</surname> <given-names>A.</given-names></name> <name><surname>Verma</surname> <given-names>V.</given-names></name> <name><surname>Kawaguchi</surname> <given-names>K.</given-names></name> <name><surname>Matyasko</surname> <given-names>A.</given-names></name> <name><surname>Khosla</surname> <given-names>S.</given-names></name> <name><surname>Kannala</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Interpolated adversarial training: achieving robust neural networks without sacrificing too much accuracy</article-title>. <source>Neural Netw.</source> <volume>154</volume>, <fpage>218</fpage>&#x2013;<lpage>233</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neunet.2022.07.012</pub-id>, PMID: <pub-id pub-id-type="pmid">35930854</pub-id></citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Larochelle</surname> <given-names>H.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Louradour</surname> <given-names>J.</given-names></name> <name><surname>Lamblin</surname> <given-names>P.</given-names></name></person-group> (<year>2009</year>). <article-title>Exploring strategies for training deep neural networks</article-title>. <source>J. Mach. Learn. Res.</source> <volume>1</volume>, <fpage>1</fpage>&#x2013;<lpage>40</lpage>. doi: <pub-id pub-id-type="doi">10.1145/1577069.1577070</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>LeCun</surname> <given-names>Y.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>Nature</source> <volume>521</volume>, <fpage>436</fpage>&#x2013;<lpage>444</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature14539</pub-id></citation></ref>
<ref id="ref33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Madani</surname> <given-names>K.</given-names></name> <name><surname>Kachurka</surname> <given-names>V.</given-names></name> <name><surname>Sabourin</surname> <given-names>C.</given-names></name> <name><surname>Amarger</surname> <given-names>V.</given-names></name> <name><surname>Golovko</surname> <given-names>V.</given-names></name> <name><surname>Rossi</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>A human-like visual-attention-based artificial vision system for wildland firefighting assistance</article-title>. <source>Appl. Intell.</source> <volume>48</volume>, <fpage>2157</fpage>&#x2013;<lpage>2179</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10489-017-1053-6</pub-id></citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Menezes</surname> <given-names>A.</given-names></name> <name><surname>de Moura</surname> <given-names>G.</given-names></name> <name><surname>Alves</surname> <given-names>C.</given-names></name> <name><surname>de Carvalho</surname> <given-names>A. C. P. L. F.</given-names></name></person-group> (<year>2023</year>). <article-title>Continual object detection: a review of definitions, strategies, and challenges</article-title>. <source>Neural Netw.</source> <volume>161</volume>, <fpage>476</fpage>&#x2013;<lpage>493</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neunet.2023.01.041</pub-id>, PMID: <pub-id pub-id-type="pmid">36805263</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Mikolov</surname> <given-names>T.</given-names></name> <name><surname>Deoras</surname> <given-names>A.</given-names></name> <name><surname>Povey</surname> <given-names>D.</given-names></name> <name><surname>Burget</surname> <given-names>L.</given-names></name> <name><surname>Cernocky</surname> <given-names>J.</given-names></name></person-group> (<year>2011</year>). &#x201C;<article-title>Strategies for training large scale neural network language models</article-title>&#x201D; in <source>Automatic Speech Recognition and Understanding</source>, <fpage>195</fpage>&#x2013;<lpage>201</lpage>.</citation></ref>
<ref id="ref36"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Nair</surname> <given-names>V.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2010</year>). Rectified linear units improve restricted Boltzmann machines. In <italic>Proceedings of the 27th international conference on machine learning (ICML-10)</italic>, pp.807&#x2013;814.</citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nakamura</surname> <given-names>K.</given-names></name> <name><surname>Derbel</surname> <given-names>B.</given-names></name> <name><surname>Won</surname> <given-names>K.</given-names></name> <name><surname>Hong</surname> <given-names>B.</given-names></name></person-group> (<year>2021</year>). <article-title>Learning-rate annealing methods for deep neural networks</article-title>. <source>Electronics</source> <volume>10</volume>:<fpage>2029</fpage>:<fpage>2029</fpage>. doi: <pub-id pub-id-type="doi">10.3390/electronics10162029</pub-id></citation></ref>
<ref id="ref38"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Pesme</surname> <given-names>S.</given-names></name> <name><surname>Dieuleveut</surname> <given-names>A.</given-names></name> <name><surname>Flammarion</surname> <given-names>N.</given-names></name></person-group> (<year>2020</year>). On convergence-diagnostic based step sizes for stochastic gradient descent. In: <italic>Proceedings of the international conference on machine learning, ICML</italic>, 119, pp. 7641&#x2013;7651.</citation></ref>
<ref id="ref39"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Pouyanfar</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>S. C.</given-names></name></person-group> (<year>2017</year>). T-LRA: trend-based learning rate annealing for deep neural networks. In <italic>Proceedings of the 2017 IEEE third international conference on multimedia big data (BigMM)</italic>, Laguna Hills, CA, USA; pp. 50&#x2013;57.</citation></ref>
<ref id="ref40"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Ruder</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). An overview of gradient descent optimization algorithms, Available at: <ext-link xlink:href="https://arxiv.org/abs/1609.04747" ext-link-type="uri">https://arxiv.org/abs/1609.04747</ext-link>.</citation></ref>
<ref id="ref41"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Schaul</surname> <given-names>T.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>LeCun</surname> <given-names>Y.</given-names></name></person-group> (<year>2013</year>). No more pesky learning rates. In <italic>Proceedings of the international conference on machine learning (ICML-2013)</italic>, Atlanta, GA, USA; pp. 343&#x2013;351.</citation></ref>
<ref id="ref42"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Scholz</surname> <given-names>M.</given-names></name> <name><surname>Fraunholz</surname> <given-names>M.</given-names></name> <name><surname>Selbig</surname> <given-names>J.</given-names></name></person-group> (<year>2008</year>). <source>Nonlinear principal component analysis: Neural network models and applications, in principal manifolds for data visualization and dimension reduction</source>, <publisher-name>Springer</publisher-name> <publisher-loc>Berlin Heidelberg</publisher-loc>, <fpage>44</fpage>&#x2013;<lpage>67</lpage>.</citation></ref>
<ref id="ref43"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>L. N.</given-names></name></person-group> (<year>2017</year>). Cyclical learning rates for training neural networks. In: <italic>IEEE Winter Conference on Applications of Computer Vision (WACV)</italic>.</citation></ref>
<ref id="ref44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Takase</surname> <given-names>T.</given-names></name> <name><surname>Oyama</surname> <given-names>S.</given-names></name> <name><surname>Kurihara</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>Effective neural network training with adaptive learning rate based on training loss</article-title>. <source>Neural Netw.</source> <volume>101</volume>, <fpage>68</fpage>&#x2013;<lpage>78</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neunet.2018.01.016</pub-id>, PMID: <pub-id pub-id-type="pmid">29494873</pub-id></citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaswani</surname> <given-names>S.</given-names></name> <name><surname>Mishkin</surname> <given-names>A.</given-names></name> <name><surname>Laradji</surname> <given-names>I.</given-names></name> <name><surname>Schmidt</surname> <given-names>M.</given-names></name> <name><surname>Gidel</surname> <given-names>G.</given-names></name> <name><surname>Lacoste-Julien</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>Painless stochastic gradient: interpolation, line-search, and convergence rates</article-title>. <source>Adv. Neural Inf. Proces. Syst.</source></citation></ref>
<ref id="ref46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Verma</surname> <given-names>V.</given-names></name> <name><surname>Kawaguchi</surname> <given-names>K.</given-names></name> <name><surname>Lamb</surname> <given-names>A.</given-names></name> <name><surname>Kannala</surname> <given-names>J.</given-names></name> <name><surname>Solin</surname> <given-names>A.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Interpolation consistency training for semi-supervised learning</article-title>. <source>Neural Netw.</source> <volume>145</volume>, <fpage>90</fpage>&#x2013;<lpage>106</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neunet.2021.10.008</pub-id>, PMID: <pub-id pub-id-type="pmid">34735894</pub-id></citation></ref>
<ref id="ref47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.-J.</given-names></name> <name><surname>Gao</surname> <given-names>H.-B.</given-names></name> <name><surname>Wang</surname> <given-names>X.-H.</given-names></name> <name><surname>Zhao</surname> <given-names>S.-Y.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>X.-Q.</given-names></name></person-group> (<year>2023</year>). <article-title>Adaptive learning rate optimization algorithms with dynamic bound based on Barzilai-Borwein method</article-title>. <source>Inform. Sci.</source> <volume>634</volume>, <fpage>42</fpage>&#x2013;<lpage>54</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ins.2023.03.050</pub-id></citation></ref>
<ref id="ref48"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zeiler</surname> <given-names>M. D.</given-names></name></person-group> (<year>2012</year>). Adadelta: An adaptive learning method. ArXiv [Preprint]. arXiv:1212.5701.</citation></ref></ref-list>
</back>
</article>