<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1605706</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Effective methods and framework for energy-based local learning of deep neural networks</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Haibo</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3025373/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Yang</surname> <given-names>Bangcheng</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>He</surname> <given-names>Fucun</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhou</surname> <given-names>Fei</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Shuai</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wu</surname> <given-names>Chunpeng</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1317267/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Fan</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Chua</surname> <given-names>Yansong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/36075/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>China Nanhu Academy of Electronics and Information Technology</institution>, <addr-line>Jiaxing</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>China Electric Power Research Institute</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>State Grid Shanghai Municipal Electric Power Company</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Georgios Leontidis, University of Aberdeen, United Kingdom</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Paul Bogdan, University of Southern California, United States</p>
<p>Jangho Lee, Incheon National University, Republic of Korea</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Yansong Chua <email>caiyansong&#x00040;cnaeit.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>26</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1605706</elocation-id>
<history>
<date date-type="received">
<day>03</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Chen, Yang, He, Zhou, Chen, Wu, Li and Chua.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Chen, Yang, He, Zhou, Chen, Wu, Li and Chua</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>From a neuroscience perspective, artificial neural networks are regarded as abstract models of biological neurons, yet they rely on biologically implausible backpropagation for training. Energy-based models represent a class of brain-inspired learning frameworks that adjust system states by minimizing an energy function. Predictive coding (PC), a theoretical model within energy-based models, constructs its energy function from forward prediction errors, with optimization achieved by minimizing local layered errors. Owing to its local plasticity, PC emerges as the most promising alternative to backpropagation. However, PC face gradient explosion and vanishing challenges in deep networks with multiple layers. Gradient explosion occurs when layer-wise prediction errors are excessively large, while gradient vanishing arises when they are excessively small. To address these challenges, we propose bidirectional energy to stabilize prediction errors and mitigate gradient explosion, while using skip connections to resolve gradient vanishing problems. We also introduce a layer-adaptive learning rate (LALR) to enhance training efficiency. Our model achieves accuracies of 99.22% on MNIST, 93.78% on CIFAR-10, 83.96% on CIFAR-100, and 73.35% on Tiny ImageNet, comparable to the performance of identically structed networks trained with backprop. Finally, we developed a Jax-based framework for efficient training of energy-based models, reducing training time by half compared to PyTorch.</p></abstract>
<kwd-group>
<kwd>artificial neural network</kwd>
<kwd>biologically plausible learning rule</kwd>
<kwd>local learning</kwd>
<kwd>energy-based model</kwd>
<kwd>predictive coding</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="3"/>
<equation-count count="23"/>
<ref-count count="60"/>
<page-count count="16"/>
<word-count count="10194"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Artificial neural networks (ANNs) trained using backpropagation (backprop) have achieved remarkable advancements over the past decade. Despite this success, neuroscientists have questioned the biological plausibility of backprop. A key criticism is that biological neurons adhere to rules of accessing information only from adjacent neurons locally, whereas backprop transmits information from distant neurons layer by layer via the chain rule (<xref ref-type="bibr" rid="B10">Crick, 1989</xref>; <xref ref-type="bibr" rid="B51">Stork, 1989</xref>). This disparity has prompted researchers to explore alternative solutions based on biology, particularly bio-inspired models and learning algorithms. Some studies on the structure and function of biological neural networks. Such as the simulation modeling of neural network connection methods (<xref ref-type="bibr" rid="B55">Yang et al., 2021</xref>; <xref ref-type="bibr" rid="B23">Hoffmann et al., 2024</xref>), the topological structure and interaction of neural connections in biological neural networks (<xref ref-type="bibr" rid="B60">Znaidi et al., 2023</xref>; <xref ref-type="bibr" rid="B5">Boccato et al., 2024</xref>; <xref ref-type="bibr" rid="B48">Salova and Kov&#x000E1;cs, 2025</xref>; <xref ref-type="bibr" rid="B53">Xiao et al., 2021</xref>), the interconnection structure, self-organization and self-optimization characteristics of brain-derived neurons (<xref ref-type="bibr" rid="B57">Yin et al., 2020</xref>), all of these studies have pointed out that the local interactions of the connections in biological neural networks have adaptive adjustment characteristics, which can optimize the overall information transmission efficiency. At the same time, some &#x0201C;backprop-free&#x0201D; local learning methods that avoid global gradient transmission have been proposed. These methods aim to modify the weights of the dynamical equations by using locally available information. Such methods are usually strongly inspired by biological synaptic plasticity and give rise to various algorithms and models. These models include self-organizing maps (<xref ref-type="bibr" rid="B25">Khacef et al., 2019</xref>; <xref ref-type="bibr" rid="B22">Hirani et al., 2024</xref>; <xref ref-type="bibr" rid="B47">Sa-Couto and Wichert, 2023</xref>), hebbian learning (<xref ref-type="bibr" rid="B42">Pogodin and Latham, 2020</xref>; <xref ref-type="bibr" rid="B28">Krotov and Hopfield, 2019</xref>; <xref ref-type="bibr" rid="B38">Moraitis et al., 2022</xref>), forward-forward algorithms (<xref ref-type="bibr" rid="B21">Hinton, 2022</xref>; <xref ref-type="bibr" rid="B37">Momeni et al., 2023</xref>), feedback alignment algorithms (<xref ref-type="bibr" rid="B31">Lillicrap et al., 2016</xref>; <xref ref-type="bibr" rid="B39">N&#x000F8;kland, 2016</xref>), local error-driven (<xref ref-type="bibr" rid="B8">Cheng et al., 2024</xref>; <xref ref-type="bibr" rid="B56">Yin et al., 2023</xref>), energy-based local learning models (<xref ref-type="bibr" rid="B4">Bengio and Fischer, 2015</xref>; <xref ref-type="bibr" rid="B24">Hopfield, 1982</xref>; <xref ref-type="bibr" rid="B49">Scellier and Bengio, 2017</xref>; <xref ref-type="bibr" rid="B54">Xie and Seung, 2003</xref>).</p>
<p>Energy-based local learning models (EBLL) originate from the broader category of energy models, which view learning and inference as the minimization of an energy function defined over the states of model variables (such as inputs, outputs, or hidden states). These models must define and estimate an explicit global energy function. EBLL typically adheres to classical energy theories, such as the free energy principle or hopfield energy. During the energy minimization process, EBLL minimizes local energy through a locality principle, either hierarchically or in blocks, thereby avoiding the propagation of global energy gradients. Even under the guidance of classical energy theories, defining and estimating an energy function remains challenging in practical applications, such as hierarchical predictive coding (HPC) models (<xref ref-type="bibr" rid="B14">Friston, 2005</xref>; <xref ref-type="bibr" rid="B18">Friston and Stephan, 2007</xref>; <xref ref-type="bibr" rid="B52">Whittington and Bogacz, 2017</xref>; <xref ref-type="bibr" rid="B7">Buckley et al., 2017</xref>; <xref ref-type="bibr" rid="B34">Millidge et al., 2021</xref>) based on the free energy principle. The free energy principle (<xref ref-type="bibr" rid="B14">Friston, 2005</xref>; <xref ref-type="bibr" rid="B17">Friston et al., 2006</xref>; <xref ref-type="bibr" rid="B18">Friston and Stephan, 2007</xref>; <xref ref-type="bibr" rid="B15">Friston, 2010</xref>) is a normative theoretical framework that asserts that systems maintain a generative model and minimize a quantity called free energy to reduce the mismatch between predicted and observed sensory data. HPC (<xref ref-type="bibr" rid="B44">Rao and Ballard, 1999</xref>; <xref ref-type="bibr" rid="B13">Friston, 2003</xref>; <xref ref-type="bibr" rid="B9">Clark, 2013</xref>) is an implementation model that describes how the brain achieves perception through the minimization of local errors. After the free energy principle was proposed, predictive coding became an approximate implementation of it (<xref ref-type="bibr" rid="B7">Buckley et al., 2017</xref>; <xref ref-type="bibr" rid="B34">Millidge et al., 2021</xref>). In the free energy principle, predictive coding assumes the generative model to be a hierarchical Gaussian probabilistic model, and free energy is defined as the difference between an approximate variational posterior distribution and the true posterior distribution, which is not easy to estimate directly (<xref ref-type="bibr" rid="B16">Friston and Kiebel, 2009</xref>; <xref ref-type="bibr" rid="B50">Spratling, 2017</xref>; <xref ref-type="bibr" rid="B40">Piekarski, 2023</xref>). In most specific supervised task implementations (<xref ref-type="bibr" rid="B52">Whittington and Bogacz, 2017</xref>; <xref ref-type="bibr" rid="B11">Dold et al., 2019</xref>; <xref ref-type="bibr" rid="B46">Rosenbaum, 2022</xref>; <xref ref-type="bibr" rid="B36">Millidge et al., 2022</xref>), this expression of free energy is approximated as the sum of squared local feedforward prediction errors between layers, that is, a quadratic energy function of the squared prediction errors in a single direction. This approximate expression relies on the assumption of a Gaussian distribution. This quadratic energy function offers notable computational advantages and is widely adopted in practice. Mathematically, it is a convex function with continuous gradients and analytical derivatives, usually ensuring the existence of a unique minimum. This characteristic is particularly convenient for calculation in the process of minimizing the energy function. However, it also has significant limitations. The real perception mappings and deep network representations are often highly non-Gaussian and nonlinear. In high-dimensional spaces, this can result in substantial errors, potentially leading to instability or even divergence during the learning process. In deep networks, this manifests as gradient explosion and vanishing gradient phenomena. When the prediction error of a single internal layer in an artificial neural network is too large, it amplifies in deeper layers, leading to high energy levels and gradient explosion during the energy minimization phase. Conversely, overly small energy levels, typically caused by network depth, can impede progress toward energy minimization.</p>
<p>Recent studies (<xref ref-type="bibr" rid="B41">Pinchetti et al., 2022</xref>; <xref ref-type="bibr" rid="B26">Kinghorn et al., 2022</xref>) have demonstrated that HPC suffers from performance degradation or outright collapse during training when applied to complex or deep neural network architectures. To address these challenges, some approaches have been proposed. Both <xref ref-type="bibr" rid="B26">Kinghorn et al. (2022)</xref> and <xref ref-type="bibr" rid="B35">Millidge et al. (2023)</xref> introduce the weight-regularization, a regularization method for HPC that uses the L1 weight norm and a simple weight restriction strategy to prevent performance degradation. While regularization is generally an effective technique for improving stability, its practical reliability remains inconsistent. <xref ref-type="bibr" rid="B41">Pinchetti et al. (2022)</xref> extended the HPC based on Gaussian distribution to any probability distribution and successfully applied it to a transformer network with a single head and 128 dimensions. The energy function shifts from minimizing the numerical prediction error to minimizing the discrepancy between the predicted distribution and the true distribution, typically measured by the Kullback-Leibler (KL) divergence. However, this approach can be computationally demanding, as the KL divergence is often intractable. In many tasks, the true distribution is either unknown or cannot be analytically integrated. Therefore, this method underscores its broad applicability to any family of distributions for which an explicit analytical form of the KL divergence can be derived.</p>
<p>Taking into account the computational advantages of the energy minimization process, we still followed the energy function in the form of prediction error under the assumption of Gaussian distribution. However, we made a structured improvement to the energy function to overcome the gradient explosion and vanishing during training. This improvement was inspired by two key aspects in biology. First, we consider the biological perspective, focusing on the reciprocal interactions between feedforward and feedback connections in cortical regions (<xref ref-type="bibr" rid="B45">Rockland, 2022</xref>; <xref ref-type="bibr" rid="B2">Angelucci and Petreanu, 2023</xref>). Second, we take inspiration from machine learning&#x00027;s emulation of biological processes. For example, <xref ref-type="bibr" rid="B32">Lillicrap et al. (2020)</xref> suggested that feedback pathways primarily adjust neural activities to transmit information essential for effective multilayer learning. Consequently, <xref ref-type="bibr" rid="B31">Lillicrap et al. (2016)</xref> introduced feedback alignment (FA), a biologically plausible learning model. However, the use of fixed random feedback weights limits FA&#x00027;s effectiveness in deeper networks. Later studies extended FA through bidirectional learning, refining the reverse pathways to improve feedback transmission (<xref ref-type="bibr" rid="B1">Amit, 2019</xref>; <xref ref-type="bibr" rid="B33">Luo et al., 2017</xref>). Similarly, target propagation (<xref ref-type="bibr" rid="B3">Bengio, 2014</xref>; <xref ref-type="bibr" rid="B30">Lee et al., 2015</xref>) algorithm employs stacked auto-encoders to reverse reconstruct local representations, guiding the learning process. These prior research findings inspire the use of bidirectional free energy to mitigate gradient explosion issues in deep PC networks. Bidirectional energy functions go through the following mechanisms:</p>
<list list-type="simple">
<list-item><p>i) Feedback connections relay signals from higher-level to lower-level units, where they are processed to produce prediction errors.</p></list-item>
<list-item><p>ii) The feedback and feedforward prediction errors together create a bidirectional symmetry in energy.</p></list-item>
<list-item><p>iii) During the energy minimization process, the interplay between bidirectional prediction errors in both directions stabilizes neuronal updates, alleviating the issues of gradient explosion in HPC networks.</p></list-item>
</list>
<p>The remainder of this paper is organized as follows. Section 2 reviews HPC and its gradient explosion and vanishing problems. Section 3 presents an overview of bidirectional PC (BiPC), illustrated with a biologically inspired ANN model. This approach integrates both bottom-up and top-down predictions in each network block, ensuring stable gradient updates during energy minimization. Section 4 explores the layer-wise weight update mechanism of EBLL and introduces the layer-adaptive learning rate (LALR). By dynamically adjusting learning parameters across network layers, LALR enhances convergence speed while ensuring stability. Section 5 presents a unified energy function framework for PC and EP in a supervised learning context. We propose that the energy function under the supervised learning scenario is split into internal and external energy components. The internal energy reflects the intrinsic dynamics of the PC, while the external energy represents the impact of the loss function. We then implement a Jax-based framework (<xref ref-type="bibr" rid="B19">Frostig et al., 2018</xref>; <xref ref-type="bibr" rid="B6">Bradbury et al., 2018</xref>) for training energy-based models, reducing training time by half compared to PyTorch. Finally, Section 6 demonstrates the effectiveness of our approach through experiments, including image classification on MNIST, CIFAR10, CIFAR100, and Tiny ImageNet. The results show that our framework enables reliable learning within deep ANNs using EBLL, achieving accuracy similar to that of backprop under identical ANN conditions.</p></sec>
<sec id="s2">
<title>2 Hierarchical predictive coding</title>
<p>The classical HPC model in the visual cortex is an unsupervised learning framework (<xref ref-type="bibr" rid="B44">Rao and Ballard, 1999</xref>), where top-down processing generates predictions, with feedback pathways transmitting predictions from the activities of higher-level to lower-level units. Prediction errors are processed bottom-up, with the feedforward pathways carrying the residuals between the predictions and the actual activities (<xref ref-type="bibr" rid="B44">Rao and Ballard, 1999</xref>). HPC efficiently encodes input data by iteratively looping and locally predicting and correcting input signals through its hierarchical structure. However, applied to supervised machine learning tasks in an ANN, HPC functions in a manner contrary to its theoretical model described above (<xref ref-type="bibr" rid="B36">Millidge et al., 2022</xref>; <xref ref-type="bibr" rid="B46">Rosenbaum, 2022</xref>). The data <italic>X</italic> is assigned to layer 0 at the bottom, and labels <italic>Y</italic> are assigned to layer <italic>L</italic>&#x0002B;1 at the top. Predictions are made through the feedforward process, while prediction errors at each level adjust unit activities and parameters in a backward direction. The detailed process is outlined as follows.</p>
<p>We consider an <italic>L</italic>-layers feedforward network with unit states <italic>V</italic>, where <italic>v</italic><sub><italic>i</italic></sub> &#x02208; <italic>V</italic> denotes the latent states of the <italic>ith</italic> layer, with <italic>v</italic><sub>0</sub> &#x0003D; <italic>X</italic> and <italic>v</italic><sub><italic>L</italic>&#x0002B;1</sub> &#x0003D; <italic>Y</italic>. We refer to <xref ref-type="bibr" rid="B46">Rosenbaum (2022)</xref> for the state initialization protocol, where the initial forward pass is used as initial values. In a supervised learning context, the output of the network must converge to the label <italic>Y</italic>. The prediction errors for each layer are defined as follows</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>g</italic><sub><italic>i</italic></sub>(<italic>v</italic><sub><italic>i</italic>&#x02212;1</sub>; &#x003B8;<sub><italic>i</italic></sub>) implies the <italic>ith</italic> layer function applied to <italic>v</italic><sub><italic>i</italic>&#x02212;1</sub>, yielding the feedforward prediction <inline-formula><mml:math id="M3"><mml:mover accent="true"><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>.</p>
<p>Ultimately, supervised learning HPC optimizes a global energy function <italic>F</italic>, which includes internal layer-wise prediction errors and a loss function applied to the set of output units (<xref ref-type="bibr" rid="B11">Dold et al., 2019</xref>).</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mi>C</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>C</italic> indicates the mean squared error between the output prediction and target behavior (<xref ref-type="bibr" rid="B11">Dold et al., 2019</xref>). Both unit states and parameter dynamics of the network can be derived as a gradient descent on the energy function <italic>F</italic>. Therefore, <italic>F</italic> can also be interpreted as the global objective function of the network (<xref ref-type="bibr" rid="B36">Millidge et al., 2022</xref>).</p>
<p>At each iteration, the network states are updated as follows: <italic>v</italic><sub><italic>i</italic></sub> &#x0003D; <italic>v</italic><sub><italic>i</italic></sub>&#x02212;&#x003B7;<sub><italic>v</italic></sub><italic>dv</italic><sub><italic>i</italic></sub>, where &#x003B7;<sub><italic>v</italic></sub> refers to the step rate for <italic>v</italic>, and <italic>dv</italic><sub><italic>i</italic></sub> is expressed as:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>After sufficient iterations, <italic>F</italic> eventually converges to its equilibrium point <italic>F</italic><sub><italic>min</italic></sub>. At this point, the parameters &#x003B8;<sub><italic>i</italic></sub> are updated as &#x003B8;<sub><italic>i</italic></sub> &#x0003D; &#x003B8;<sub><italic>i</italic></sub>&#x02212;&#x003B7;<sub>&#x003B8;</sub><italic>d&#x003B8;</italic><sub><italic>i</italic></sub>, where &#x003B7;<sub>&#x003B8;</sub> implies the step size for updating &#x003B8;. The formula for <italic>d&#x003B8;</italic><sub><italic>i</italic></sub> is given by:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Although supervised HPC adheres to local updates, some studies suggest that HPC in supervised learning approximates backprop. This indicates that HPC possesses notable potential. However, it also faces gradient explosion and vanishing. Gradient explosion occurs when a large &#x003F5;<sub><italic>i</italic></sub> amplifies subsequent errors. In the process of minimizing the energy function, the computation of the <italic>dv</italic><sub><italic>i</italic></sub>, which is derived from <xref ref-type="disp-formula" rid="E3">Equation 3</xref>, depends on two terms: the first term &#x003F5;<sub><italic>i</italic></sub>, and the second term <inline-formula><mml:math id="M9"><mml:msub><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></inline-formula>. if &#x003F5;<sub><italic>i</italic></sub> is excessively large, even a moderate second term may still result in an overly large gradient. Furthermore, updating the state <italic>v</italic><sub><italic>i</italic></sub>, expressed as <italic>v</italic><sub><italic>i</italic></sub> &#x0003D; <italic>v</italic><sub><italic>i</italic></sub>&#x02212;&#x003B7;<sub><italic>v</italic></sub><italic>dv</italic><sub><italic>i</italic></sub>, can lead to significant changes in <italic>v</italic><sub><italic>i</italic></sub> when <inline-formula><mml:math id="M10"><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></inline-formula> or the learning rate &#x003B7;<sub><italic>v</italic></sub> is excessively large. As the hierarchical network propagates forward, such drastic changes in <italic>v</italic><sub><italic>i</italic></sub> influence the prediction of the next layer, given by <inline-formula><mml:math id="M11"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, consequently increasing &#x003F5;<sub><italic>i</italic>&#x0002B;1</sub>. This effect becomes amplified in deep networks. In cases where the network depth is substantial, errors and gradients accumulate progressively during inter-layer propagation, ultimately leading to gradient explosion. Conversely, the same mechanism can contribute to gradient vanishing.</p></sec>
<sec id="s3">
<title>3 Bidirectional predictive coding</title>
<p>Here, we introduce a novel, biologically plausible BiPC model. As described in Section 2, supervised HPC relies only on feedforward prediction error driven. When the prediction error in one layer is excessively large or small, the cumulative effect of forward propagation through the hierarchical network frequently results in severe explosion or vanishing of the gradient, thereby constraining its effectiveness for complex tasks. The BiPC model overcomes these limitations by incorporating bidirectional error propagation. As noted in Section 1, this bidirectional architecture is inspired by the reciprocal connectivity observed in cortical neural networks, exemplified by the interplay between feedforward pathways from the primary visual cortex (e.g., V1) to extrastriate cortex (e.g., V2, V3, V4) and feedback pathways from extrastriate cortex back to V1. These feedback signals modulate feedforward inputs&#x02013;either amplifying or suppressing them&#x02013;to refine visual perception. Grounded in a biologically inspired artificial neural network (ANN), the BiPC model features a feedback prediction pathway from higher to lower representations, enabling symmetric modulation between feedforward and feedback prediction errors at the same level. This symmetry suppresses excessive error signals, mitigating gradient explosion. Furthermore, skip connections are employed to strengthen the flow of gradients to deeper layers, addressing gradient vanishing and ensuring robust learning across the network.</p>
<p>We model a network of nodes and edges designed to simulate the cortical structure. Each node represents a cortical area in the brain, such as V1 or V2, and encodes the activation values of all units at a given time, which are subsequently transmitted to the next node through edges. The bottom node captures the initial input data features, while the top node encodes advanced features for label prediction.</p>
<p>The edges between these nodes represent four distinct connectivity patterns found in cortical regions: feedforward connection and feedback connection, skip connection, and recurrent connection. These connections work together to generate state predictions within cortical regions:</p>
<list list-type="bullet">
<list-item><p>The bottom-up feedforward based on inference prediction circuit extracts high-level representations from input signals for decision-making.</p></list-item>
<list-item><p>The top-down feedback based on the generative prediction circuit generates estimates grounded on high-level representations from the inference circuit.</p></list-item>
<list-item><p>The recurrent connections enable bidirectional propagation of node states at each level, from left to right and vice versa.</p></list-item>
</list>
<sec>
<title>3.1 Feedforward prediction</title>
<p>We demonstrate the inference and learning process of the BiPC model using the states of an intermediate node <italic>v</italic><sub><italic>m, t</italic></sub> as a representative example.</p>
<p>As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, the feedforward state predictions for the <italic>mth</italic> node require three input components: the first component is the output of the node states <italic>v</italic><sub><italic>m</italic>&#x02212;1, <italic>t</italic>&#x02212;1</sub> from the previous node processed through the feedforward connection; the second component is the output of the node states <italic>v</italic><sub><italic>m</italic>&#x02212;2, <italic>t</italic>&#x02212;1</sub> from a lower node, processed through the skip connection; and the third component is the node&#x00027;s own output from the previous time step.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>The comparison between the traditional HPC and our proposed BiPC model for supervised tasks. <bold>(a)</bold> In the HPC model, only bottom-up predictions (represented by black arrows) are present. Forward prediction errors (denoted by orange triangles) are computed by comparing these predictions with the actual unit states, and the derived error signals serve as feedback (shown as red arrows) to modify the unit states. <bold>(b)</bold> The BiPC model features symmetric connectivity, integrating both feedforward and feedback prediction routes. This results in the generation of both feedforward and feedback prediction errors (illustrated by orange and purple triangles, respectively), along with the inclusion of recurrent and skip connections. <bold>(c)</bold> An in-depth depiction of the evolution of unit states over time. The red dashed box highlights the unit state <italic>v</italic><sub><italic>m, t</italic></sub>, while the semi-transparent modules represent components not directly linked to <italic>v</italic><sub><italic>m, t</italic></sub>. On the left, both feedforward and feedback predictions for <italic>v</italic><sub><italic>m, t</italic></sub> are presented, producing feedforward prediction errors <inline-formula><mml:math id="M12"><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> (orange triangle) and feedback prediction errors <inline-formula><mml:math id="M13"><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> (purple triangle). On the right, the energy minimization phase is demonstrated, where the bidirectional prediction errors are employed to modify the unit states.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1605706-g0001.tif">
<alt-text>Diagram illustrating a multi-layer neural network model with three panels labeled (a), (b), and (c). Panel (a) shows a simple feedforward process with bottom-up predictions. Panel (b) includes feedback loops for error correction. Panel (c) demonstrates a sequence of predictions over time (t-1 to t&#x0002B;1), with arrows indicating prediction and error flows. Legends denote neuron state sets, forward and backward prediction errors, and various prediction directions. </alt-text>
</graphic>
</fig>
<p>Let <inline-formula><mml:math id="M14"><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> signifies the feedforward state predictions of the <italic>mth</italic> node at time <italic>t</italic> which is expressed as:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M15"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:msubsup><mml:mover accent='true'><mml:mi>v</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mi>f</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mover><mml:mover><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mi>&#x003B8;</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mrow><mml:mo stretchy='true'>&#x0FE37;</mml:mo></mml:mover><mml:mrow><mml:mtext>bottom-up&#x000A0;connection</mml:mtext></mml:mrow></mml:mover></mml:mrow></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>+</mml:mo><mml:mover><mml:mover><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mi>&#x003B8;</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mrow><mml:mo stretchy='true'>&#x0FE37;</mml:mo></mml:mover><mml:mrow><mml:mtext>skip&#x000A0;connection</mml:mtext></mml:mrow></mml:mover></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>+</mml:mo><mml:mover><mml:mover><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy='true'>&#x0FE37;</mml:mo></mml:mover><mml:mrow><mml:mtext>self-excitation</mml:mtext></mml:mrow></mml:mover><mml:mo>;</mml:mo><mml:msub><mml:mi>&#x003B8;</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>g</italic><sub><italic>m</italic>&#x02212;1, <italic>m</italic></sub> indicates the connection function from the lower node <italic>m</italic>&#x02212;1 to the higher node <italic>m</italic>, <italic>g</italic><sub><italic>m, m</italic></sub> implies the recurrent connection function for the node <italic>m</italic>, and <italic>g</italic><sub><italic>m</italic>&#x02212;2, <italic>m</italic></sub> denotes a skip-connection function from the lower node <italic>m</italic>&#x02212;2 to the higher-level node <italic>m</italic>. The predicted state is then compared to the actual activity of <italic>b</italic><sub><italic>m</italic></sub> at time step <italic>t</italic>, yielding the forward prediction errors, <inline-formula><mml:math id="M17"><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>.</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M18"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>.</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></sec>
<sec>
<title>3.2 Feedback prediction</title>
<p>Previous research has examined some methods for modeling the feedback pathways to simulate the brain feedback. One approach involves creating a secondary feedback network (<xref ref-type="bibr" rid="B20">Hinton, 2003</xref>; <xref ref-type="bibr" rid="B54">Xie and Seung, 2003</xref>), often requiring the presence of reverse connections that mirror the forward connections. In this study, we design the feedback pathways to produce feedback predictions of node states, which are integrated with feedforward prediction errors to adjust these states. This process demands symmetry between the feedback and forward prediction errors. Consider <italic>mth</italic> node as an example, according to the principle of symmetry, there should be three feedback connections symmetrical to each feedforward, skip, and recurrent connection. We exclude higher-to-lower-level skip connections due to significant information loss during the transfer, which impedes the accurate recovery of lower-level representations. Two branches are used to generate the feedback state prediction comprising <inline-formula><mml:math id="M19"><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>. One branch connects <italic>v</italic><sub><italic>m</italic>&#x0002B;1, <italic>t</italic>&#x0002B;1</sub> to <italic>v</italic><sub><italic>m, t</italic></sub>, with its feedback prediction comprising <inline-formula><mml:math id="M20"><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M21"><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>. The feedback prediction representation <inline-formula><mml:math id="M22"><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and its errors <inline-formula><mml:math id="M23"><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> are defined as follows:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mover class="msup"><mml:mrow><mml:mover accent="false"><mml:mrow><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>&#x0FE37;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">top-down connection</mml:mtext></mml:mrow></mml:mover><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mover class="msup"><mml:mrow><mml:mover accent="false"><mml:mrow><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x022A4;</mml:mo></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>&#x0FE37;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">recurrent connection</mml:mtext></mml:mrow></mml:mover><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E8"><label>(8)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>z</italic><sub><italic>m</italic>&#x0002B;1, <italic>m</italic></sub> refers to the feedback function from the high-level node <italic>m</italic>&#x0002B;1 to low-level node <italic>m</italic>, and <italic>z</italic><sub><italic>m, m</italic></sub> indicates its self-recurrent feedback function. In supervised ANN tasks, the feedforward function typically handles feature extraction and downsampling operations, while the feedback function manages signal reconstruction and upsampling. The forward encoding uses convolutional operations, while the reverse decoding employs transposed convolutions. The feedback function applies transposed convolutions with the transpose of the feedforward parameters.</p></sec>
<sec>
<title>3.3 Enery function</title>
<p>The energy equation can be expressed in <xref ref-type="disp-formula" rid="E9">Equation 9</xref>, representing the sum of two distinct PC components&#x02013;the feedforward and the feedback pathways:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>C</italic><sup><italic>f</italic></sup> indicates the feedforward loss function, specifically using cross-entropy loss for the supervised discrimination task, and <italic>C</italic><sup><italic>b</italic></sup> signifies the feedback loss function, employing mean squared error loss.</p>
<p>However, we found that this formulation often causes gradient explosion in deep networks. Since both feedforward and feedback prediction errors are squared and positive, large errors in one are not sufficiently controlled by the other, leading to a potential gradient explosion.</p>
<p>Therefore, we propose a revised energy function, which will be applied consistently throughout this study:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>f</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><xref ref-type="disp-formula" rid="E10">Equation 10</xref> mitigates gradient explosion by balancing positive and negative cancellation of feedforward and feedback prediction errors.</p>
<p>With the energy function established, the node states are updated based on the energy function using the following formula. First, the states <italic>v</italic> are updated via gradient descent based on the energy function <italic>F</italic>. Subsequently, parameter updates are performed at the minimum energy <italic>F</italic><sub>min</sub>. Both error directions are used for updating states and parameters to maintain stability during the inference and learning process, as shown in <xref ref-type="disp-formula" rid="E10">Equation 10</xref>:</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E12"><label>(12)</label><mml:math id="M29"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">min</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B7;<sub><italic>v</italic></sub> and &#x003B7;<sub>&#x003B8;</sub> refer to the step size for states and parameters update, respectively, and &#x003B8;<sub><italic>m, i</italic></sub> implies the parameters from node <italic>m</italic> to other nodes or itself.</p></sec></sec>
<sec id="s4">
<title>4 Layer-adaptive learning rate</title>
<p>DNNs are typically trained using methods such as stochastic gradient descent (SGD), which apply a fixed global learning rate (LR) across all layers. However, a fixed LR can cause inefficiencies and instabilities during training. Some adaptive update rules like AdaGrad (<xref ref-type="bibr" rid="B12">Duchi et al., 2011</xref>) and Adam (<xref ref-type="bibr" rid="B27">Kingma and Ba, 2014</xref>) adjust the global LR to mitigate these issues. However, these methods remain suboptimal for EBLL in hierarchical networks, where layer parameter updates driven by each layer&#x00027;s own parameter variations (<xref ref-type="disp-formula" rid="E4">Equation 4</xref>). Given the significant variation in gradient dynamics across layers in such models, a global LR fails to adequately address the distinct needs of each layer, resulting in inefficient convergence and potential training instability.</p>
<p>To demonstrate this point, we calculated the ratio of the model&#x00027;s weight norm based on the product of each node&#x00027;s gradient norm and the global LR. This ratio quantifies weight changes in individual nodes relative to the entire weight space. As illustrated in <xref ref-type="fig" rid="F2">Figure 2a</xref>, significant variations in weight changes are observed across different nodes. As noted by (<xref ref-type="bibr" rid="B58">You et al. 2017</xref>), (<xref ref-type="bibr" rid="B59">2020</xref>), an excessively large LR can cause parameter updates across the entire parameter space to become too large, risking divergence. This observation is evident when the ratio falls below 1, a result confirmed in our experiment and depicted in <xref ref-type="fig" rid="F2">Figures 2b</xref>, <xref ref-type="fig" rid="F2">c</xref>.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>The impact of the LALR method. <bold>(a)</bold> The ratio of weight to gradient norms across various nodes in the model, with the y-axis showing the ratio &#x02225;<italic>w</italic>&#x02225;<sub>2</sub>/(&#x003B7;&#x02225;<italic>dw</italic>&#x02225;<sub>2</sub>), and the x-axis representing different nodes. The solid lines depict the ratios for both HPC and BiPC models during the 20<italic>th</italic> epoch of normal training, while the dashed line indicates the ratio at the 30<italic>th</italic> epoch when the gradient explosion took place. The training losses of <bold>(b)</bold> HPC and <bold>(c)</bold> BiPC on the CIFAR10 dataset before and after incorporating LALR. The blue and orange lines denote Adam and LALR optimizers, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1605706-g0002.tif">
<alt-text>Three graphs illustrate different data relations. (a) A line graph with &#x02018;node number&#x00027; on the x-axis and &#x02018;ratio&#x00027; on the y-axis shows HPC_normal, BiPC_normal, and HPC_explosion trends. (b) A line graph with &#x02018;Epoch&#x00027; on the x-axis and &#x02018;Loss&#x00027; on the y-axis compares HPC and HPC(LALR), showing distinct loss trends. (c) A line graph with &#x02018;Epoch&#x00027; on the x-axis and &#x02018;Loss&#x00027; on the y-axis compares BiPC and BiPC(LALR), showing decreasing loss trends over time. </alt-text>
</graphic>
</fig>
<p>To resolve this issue, we propose adapting the LR for each layer based on its parameter changes relative to the entire parameter space, aiming to enhance training stability. This approach forms the foundation of our Layer-Adaptive Learning Rate Optimization (LALR) algorithm. LALR introduces the layer-wise LR (&#x003B7;<sub>&#x003B8;, <italic>i</italic></sub>), derived from a global LR (&#x003B7;<sub>&#x003B8;</sub>) and scaled to ensure stable and balanced updates across the model. The key idea is to normalize the parameter update magnitude of each layer to match the average update magnitude across the model&#x00027;s entire parameter space, thereby mitigating disparities in parameter change amplitudes among layers. To achieve this, we first quantify the layer-wise parameter update magnitude (&#x00394;&#x003B8;<sub><italic>i</italic></sub>) and average update magnitude of all model parameters (<inline-formula><mml:math id="M30"><mml:mtext>&#x00394;</mml:mtext><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula>).</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M31"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext>&#x00394;</mml:mtext><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mo>&#x02225;</mml:mo><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x02225;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E14"><label>(14)</label><mml:math id="M32"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext>&#x00394;</mml:mtext><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mo>&#x02225;</mml:mo><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow><mml:mo>&#x02211;</mml:mo><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x02225;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>We employ the harmonic mean function <inline-formula><mml:math id="M33"><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow></mml:math></inline-formula> to measure the update magnitude across the model&#x00027;s parameter space. This choice is motivated by the harmonic mean&#x00027;s reduced sensitivity to extreme values, which ensures a more robust estimation of the gradient behavior across all model parameters. Finally, we adjust the layer-wise learning rate &#x003B7;<sub>&#x003B8;, <italic>i</italic></sub> based on the relative balance between local parameter gradient updates (&#x00394;&#x003B8;<sub><italic>i</italic></sub>) and global parameter gradient updates (<inline-formula><mml:math id="M34"><mml:mtext>&#x00394;</mml:mtext><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula>) (see <xref ref-type="disp-formula" rid="E15">Equations 15</xref>, <xref ref-type="disp-formula" rid="E16">16</xref>).</p>
<disp-formula id="E15"><label>(15)</label><mml:math id="M35"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mtext>&#x00394;</mml:mtext><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mtext>&#x00394;</mml:mtext><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E16"><label>(16)</label><mml:math id="M36"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B7;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mo>&#x02225;</mml:mo><mml:mrow><mml:mi mathvariant="script">S</mml:mi></mml:mrow><mml:mo>&#x02211;</mml:mo><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x02225;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>&#x02225;</mml:mo><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x02225;</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The advantage of this local-global relative balance update strategy is twofold: it ensures that layers with smaller gradients receive larger layer-specific learning rates, allowing sufficient updates to local layer parameters, and that layers with larger gradients receive smaller layer-specific learning rates, preventing excessive updates in any single layer that might destabilize training. This dual benefit improves overall optimization efficiency while maintaining stability across various gradient magnitudes. As shown in <xref ref-type="fig" rid="F2">Figures 2b</xref>, <xref ref-type="fig" rid="F2">c</xref>, applying LALR significantly stabilizes the training process of both HPC and BiPC. However, LALR does not fully eliminate the risk of divergence in HPC, with occasional divergence emerging in the later stages of training. Nevertheless, it provides a notable improvement compared to HPC with standard SGD.</p></sec>
<sec id="s5">
<title>5 Energy-based framework</title>
<p>This section introduces an energy-based framework that integrates PC and equilibrium propagation (EP). EP (<xref ref-type="bibr" rid="B49">Scellier and Bengio, 2017</xref>), a key method in EBLL, uses an energy function combining Hopfield energy and output loss. In the first phase, EP solely minimizes the Hopfield energy, which includes the unit states of all network nodes, guiding the model to a steady state and generating a prediction. In the second phase, output loss is added to direct the model toward the correct target. Resultantly, the energy, including the output loss, is minimized again, causing a new steady state. Once this state is achieved, the model&#x00027;s weights are updated based on the difference between the energy gradients at the initial and second steady states.</p>
<p>As outlined in PC and EP, the EBLL method consists of two distinct phases: first, it adjusts the model&#x00027;s states to minimize energy; second, it updates the network parameters once the energy reaches its minimum. This is quite different from backpropagation. Backpropagation has only one parameter adjustment stage, that is, the output layer error is adjusted by the chain rule to propagate backward layer by layer to update the learnable parameters (<xref ref-type="fig" rid="F3">Figure 3a</xref>).</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p><bold>(a)</bold> In backprop, the error is transmitted step-by-step from the output layer to the input layer, with parameters updating based on the globally propagated errors. The green rectangle represents a layer function containing trainable parameters. <bold>(b)</bold> In the EBLL framework, node states are defined at the nodes, while parameters are situated on the edges. The total network energy consists of internal energy from the hidden states and external energy related to the output states. Both states and parameters can be updated locally based on the energy. <bold>(c)</bold> In the EBLL framework, the initial state is determined by traversing the adjacency matrix, and the network energy is computed using the energy function and the initial state. Following this process, the inference phase is performed, followed by the learning phase, to fully execute the EBLL method.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1605706-g0003.tif">
<alt-text>(a) Diagram labeled &#x0201C;Backprop&#x0201D; showing a neural network training process with nodes v0 to vout connected through functions and derivatives for error calculation. (b) Diagram labeled &#x0201C;Energy-based Local Learning&#x0201D; illustrating an improved method with node states and energy function F &#x0003D; E({vi})&#x0002B;&#x003B2;C(vout, y), highlighting inference and learning phases. (c) Flowchart detailing network creation and training process using Jax&#x00027;s JIT acceleration, divided into inference and learning phases, concluding when conditions Fmin or max iterations are met. </alt-text>
</graphic>
</fig>
<p>To effectively support EBLL, we have redefined the energy-based framework in terms of both energy form and network structure. As outlined in Section 2, any EBLL energy consists of two components:</p>
<p>Internal energy, which represents the energy contribution from all layers of the network except the output layer. It encapsulates the inherent dynamics of the network&#x00027;s internal states in the absence of influence from external supervisory signals. External energy corresponds to the supervised loss. Therefore, our framework defines the energy function as follows:</p>
<disp-formula id="E17"><label>(17)</label><mml:math id="M37"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mo>=</mml:mo><mml:mi>E</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B2;</mml:mi><mml:mi>C</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>E</italic> and <italic>C</italic> indicate the internal and external energy, respectively, <italic>E</italic> in EP implies a kind of Hopfield energy, defined as <inline-formula><mml:math id="M38"><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x02260;</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:munder><mml:mi>&#x003C1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mi>&#x003C1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mi>&#x003C1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, and <italic>E</italic> in PC signifies the sum of squares of the prediction error, calculated from the actual and estimated node states of the network, Here, &#x003B2; &#x02208; [0, 1] refers to a scaling factor used to balance the influence of internal energy and external energy. Specifically, When &#x003B2; &#x0003D; 0, the influence of external energy is eliminated, and the model operates in a fully free phase, stabilizing solely based on its internal dynamics. When &#x003B2; &#x02208; (0, 1), the model becomes subject to constraints from external labels at the output layer, a requirement essential for supervised tasks. The target loss function, acting as external energy, drives the model to re-establish a balance between maintaining intrinsic state stability and achieving the target objective. When &#x003B2; &#x0003D; 1, the weights of internal and external energies are equal, maximizing the external influence while preserving the integrity of the internal structure without overwhelming.</p>
<p>The proposed network structure and energy form enable any EBLL to operate within the framework in two stages: inference and learning stages. During the inference stage, as expressed in <xref ref-type="disp-formula" rid="E18">Equation 18</xref>, the network achieves equilibrium by adjusting <italic>v</italic><sub><italic>i</italic></sub> to minimize <italic>F</italic>. Once equilibrium is reached, the learning stage as formulated in <xref ref-type="disp-formula" rid="E19">Equation 19</xref> optimizes synaptic weights by adjusting &#x003B8;<sub><italic>i</italic></sub> to further minimize <italic>F</italic>. Notably, this approach ensures that the network dynamics naturally align with the gradient direction of the target losses.</p>
<disp-formula id="E18"><label>(18)</label><mml:math id="M39"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">inference</mml:mtext><mml:mo>:</mml:mo><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E19"><label>(19)</label><mml:math id="M40"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">learning</mml:mtext><mml:mo>:</mml:mo><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x02202;</mml:mi><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The network is constructed using nodes and edges as fundamental units, where nodes represent states and edges denote parameterized mapping functions. This design allows the framework to accommodate networks with arbitrary topologies, as represented in <xref ref-type="fig" rid="F3">Figure 3b</xref>. An adjacency matrix is defined to record the indices of edges connecting nodes.</p>
<p>The EBLL framework is implemented on the Jax backend&#x02013;a Python library developed by Google designed for high-performance array computation and program transformation (<xref ref-type="bibr" rid="B19">Frostig et al., 2018</xref>). Within this framework (<xref ref-type="fig" rid="F3">Figure 3c</xref>), PC or EP sequentially performs inference and learning phases to train the model. During the inference phase, initial states and energy are computed by traversing the adjacency matrix, followed by iterative state updates until energy minimization. In the learning phase, local computations on the minimized energy enable parallel parameter updates. By utilizing JAX&#x00027;s Just-in-Time (JIT) technology (<xref ref-type="bibr" rid="B6">Bradbury et al., 2018</xref>), operations such as automatic differentiation of any order&#x02013;including those expressed in <xref ref-type="disp-formula" rid="E18">Equations 18</xref>, <xref ref-type="disp-formula" rid="E19">19</xref> are efficiently compiled. This process converts numerical computations in the prediction process into an optimized machine code at runtime using advanced tracing and XLA compilers, as demonstrated in <xref ref-type="supplementary-material" rid="SM1">Appendices A</xref> and <xref ref-type="supplementary-material" rid="SM1">B</xref>.</p></sec>
<sec id="s6">
<title>6 Experiment</title>
<p>Our methods are trained and tested for object recognition using specific datasets and networks, with performance compared against baselines. All experiments are conducted within our energy-based framework.</p>
<sec>
<title>6.1 Experiment settings</title>
<sec>
<title>6.1.1 Datasets</title>
<sec>
<title>6.1.1.1 MNIST</title>
<p>This dataset comprises 70,000 grayscale images of handwritten digits, each measuring 28*28 pixels and representing single digits ranging from 0 to 9. The dataset is divided into training 60,000 images and 10,000 testing images. Preprocessing involves normalizing all images using channel means and standard deviations.</p></sec>
<sec>
<title>6.1.1.2 CIFAR</title>
<p>This dataset includes two main subsets: CIFAR10 and CIFAR100, containing 32*32 colored images drawn from 10 and 100 classes, respectively. Each subset comprises 50,000 training images and 10,000 testing images. Preprocessing involves data normalization and augmentation techniques such as flipping and random cropping.</p></sec>
<sec>
<title>6.1.1.3 Tiny ImageNet</title>
<p>A curated subset of the larger ImageNet dataset, Tiny ImageNet consists of 100,000 color images at a resolution of 64*64 pixels. The dataset features 200 distinct classes, each comprising 500 training images, 50 validation images, and 50 test images.</p></sec></sec>
<sec>
<title>6.1.2 Network architecture</title>
<p>We use network architectures with different spatial and temporal complexity: hierarchical feedforward network and skip connection recurrent network, as represented in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Network configuration.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Architecture</bold></th>
<th valign="top" align="center"><bold>Hierarchical feedforward network</bold></th>
<th valign="top" align="center" colspan="2"><bold>Skip connection recurrent network</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td/>
<td valign="top" align="center"><bold>Simple</bold></td>
<td valign="top" align="center"><bold>Complex</bold></td>
</tr> <tr>
<td valign="top" align="left">Number of nodes</td>
<td valign="top" align="center">11</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">8</td>
</tr> <tr>
<td valign="top" align="left">Time</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">5</td>
</tr> <tr>
<td valign="top" align="left">Nodes&#x00027; connections</td>
<td valign="top" align="center">conv3-128</td>
<td valign="top" align="center">conv3-64<sup>&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">conv3-32<sup>&#x0002A;&#x0002A;</sup></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">conv3-256 maxpool-2</td>
<td valign="top" align="center">conv3-128<sup>&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">conv3-64<sup>&#x0002A;&#x0002A;</sup></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">conv3-512 maxpool-2</td>
<td valign="top" align="center">conv3-128<sup>&#x0002A;</sup> conv3-256<sup>&#x0002A;&#x0002A;</sup> conv3-512<sup>&#x0002A;&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">conv1-128&#x00023; maxpool-2&#x00023; conv3-128<sup>&#x0002A;</sup> conv3-256<sup>&#x0002A;&#x0002A;</sup> conv3-1024<sup>&#x0002A;</sup><sup>&#x0002A;&#x0002A;</sup></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">conv3-120 maxpool-2</td>
<td valign="top" align="center">conv3-256<sup>&#x0002A;</sup> conv3-512<sup>&#x0002A;&#x0002A;</sup> conv3-200<sup>&#x0002A;&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">conv1-512&#x00023; conv3-512<sup>&#x0002A;</sup> conv3-1024<sup>&#x0002A;&#x0002A;</sup> conv3-256<sup>&#x0002A;&#x0002A;&#x0002A;</sup></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">conv1-256</td>
<td valign="top" align="center">conv3-512<sup>&#x0002A;</sup> conv3-200<sup>&#x0002A;&#x0002A;</sup> conv3-32<sup>&#x0002A;</sup><sup>&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">conv1-768&#x00023; maxpool-2&#x00023; conv3-768<sup>&#x0002A;</sup> conv3-256<sup>&#x0002A;&#x0002A;</sup> conv3-256<sup>&#x0002A;</sup><sup>&#x0002A;&#x0002A;</sup></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">conv1-80</td>
<td valign="top" align="center">conv3-200<sup>&#x0002A;</sup> conv3-32<sup>&#x0002A;&#x0002A;</sup></td>
<td valign="top" align="center">conv1-128&#x00023; conv3-128<sup>&#x0002A;</sup> conv3-256<sup>&#x0002A;&#x0002A;</sup></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">conv3-64</td>
<td valign="top" align="center">conv3-32<sup>&#x0002A;</sup></td>
<td valign="top" align="center">conv1-64&#x00023; maxpool-2&#x00023; conv3-64<sup>&#x0002A;</sup></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">conv3-100</td>
<td valign="top" align="center">flatten fc-10/100/200</td>
<td valign="top" align="center">flatten fc-512 fc-10/100/200</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">conv3-50 maxpool-2</td>
<td/>
<td/>
</tr>
 <tr>
<td/>
<td valign="top" align="center">fc-512</td>
<td/>
<td/>
</tr>
 <tr>
<td/>
<td valign="top" align="center">fc-10/100/200</td>
<td/>
<td/>
</tr></tbody>
</table>
<table-wrap-foot>
<p><sup>&#x0002A;</sup>Recurrent connection; <sup>&#x0002A;&#x0002A;</sup>forward connection; <sup>&#x0002A;&#x0002A;&#x0002A;</sup>skip connection; <sup>&#x00023;</sup>internal connection.</p>
</table-wrap-foot>
</table-wrap>
<sec>
<title>6.1.2.1 Hierarchical feedforward network</title>
<p>his network architecture is a feedforward convolutional neural network architecture comprising nine convolutional layers and two fully connected layers. The convolutional layers utilize 3*3 and 1*1 kernels with varying kernel counts per layer. Max pooling with a 2*2 kernel the size of the feature maps, followed by the application of the tanh activation function. Two fully connected layers follow the convolutional layers.</p></sec>
<sec>
<title>6.1.2.2 Skip connection recurrent network</title>
<p>As shown in <xref ref-type="fig" rid="F1">Figure 1b</xref>, the network has two variants based on spatial complexity: a simple version and a complex version. The simple version consists of 8 nodes, with the 0<italic>th</italic> node serving as the input. Each node has a single state, and the 0<italic>th</italic> node encodes the input to the 1<italic>th</italic> node using a convolution function, mimicking the retina&#x00027;s processing of visual input. The 1<italic>th</italic> node processes the input, repeating the signal and projecting it to the second high-level node via the convolution function. The 7<italic>th</italic> node decodes the output. Apart from the 1<italic>th</italic> and 7<italic>th</italic> nodes, each internal node has edge functions that map the state to the next high-level, time step, and cross-layer nodes, using 3*3 convolutions with varying channels. The complex version also uses eight nodes, with each internal node containing two states. Compared to the simple version, the edge functions include an additional internal state mapping function, implemented as a 1*1 convolution function. These convolution settings are inspired by CORnet (<xref ref-type="bibr" rid="B29">Kubilius et al., 2018</xref>) settings. Additionally, the nonlinear activation function employs the hyperbolic tangent (tanh) activation function in each layer.</p></sec></sec>
<sec>
<title>6.1.3 Hyper-parameter</title>
<p>The model was trained and tested using an NVIDIA A100 80G GPU device, with the remaining hyperparameters detailed in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Settings for model hyperparameters.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Parameter</bold></th>
<th valign="top" align="center"><bold>Description</bold></th>
<th valign="top" align="center"><bold>Value</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>Batchsize</italic></td>
<td valign="top" align="center">Number of samples per gradient update</td>
<td valign="top" align="center">128</td>
</tr> <tr>
<td valign="top" align="left"><italic>N</italic></td>
<td valign="top" align="center">Maximum number of iterations for inference phase</td>
<td valign="top" align="center">200</td>
</tr> <tr>
<td valign="top" align="left">&#x003B7;<sub><italic>v</italic></sub></td>
<td valign="top" align="center">LR for inference phase</td>
<td valign="top" align="center">0.01</td>
</tr> <tr>
<td valign="top" align="left">&#x003B7;<sub>&#x003B8;</sub></td>
<td valign="top" align="center">LR for learning phase</td>
<td valign="top" align="center">0.01</td>
</tr> <tr>
<td valign="top" align="left">&#x003B2;</td>
<td valign="top" align="center">Scaling factor for external energy</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left"><italic>Threshold</italic></td>
<td valign="top" align="center">Energy convergence threshold</td>
<td valign="top" align="center">1e&#x02013;7</td>
</tr></tbody>
</table>
</table-wrap></sec></sec>
<sec>
<title>6.2 Evaluation of the effectiveness of BiPC</title>
<p>In the EBLL model, the inference and learning phases are executed sequentially. During inference, gradient descent is applied to the energy function to minimize energy by adjusting the states. Once minimized, the weight gradient is computed to update the parameters. Managing gradient explosion or vanishing during inference is critical, as these issues indicate extreme energy values and directly affect weight updates in the learning phase. Proper gradient control during inference ensures stable and effective parameter optimization.</p>
<p>We examine whether the BiPC model, utilizing two distinct energy formulas expressed in <xref ref-type="disp-formula" rid="E9">Equations 9</xref>, <xref ref-type="disp-formula" rid="E10">10</xref>, can effectively mitigate gradient explosion and vanishing during inference. The gradient norm value during training serves as the primary indicator for detecting these issues. A near-zero gradient norm indicates vanishing gradients, while a sudden escalation by several orders of magnitude signals gradient explosion. Using the simple version of the skip connection recurrent network from <xref ref-type="table" rid="T1">Table 1</xref> and the CIFAR10 dataset as an example, we evaluate BiPC under varying network depths and connection configurations. The average gradient norm during the inference phase is computed. As illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>, the red line represents the simple version of the skip connection recurrent network without recurrent or skip connections, while the blue line depicts the same network without skip connections. The green line corresponds to the fully equipped skip connection recurrent network (simple version). To evaluate HPC&#x00027;s performance with gradients across varying depths and connection types, the top subplot of <xref ref-type="fig" rid="F4">Figure 4</xref> illustrates that HPC consistently experiences gradient explosion, regardless of whether connections are feedforward, combined feedforward and recurrent, or include skip connections. The middle subplot of <xref ref-type="fig" rid="F4">Figure 4</xref> demonstrates that with the energy equation in <xref ref-type="disp-formula" rid="E10">Equation 10</xref>, gradient explosion is effectively mitigated across various depths and connections. However, gradient vanishing remains an issue in shallower layers, especially without skip connections. This trend indicates that while the BiPC model resolves gradient explosion, addressing vanishing gradients requires the inclusion of skip connections. The bottom subfigure of <xref ref-type="fig" rid="F4">Figure 4</xref> reveals that the BiPC model using <xref ref-type="disp-formula" rid="E9">Equation 9</xref> experiences the gradient explosion across various depths and connections. Similarly, we evaluate HPC and BiPC learning performance on MNIST, CIFAR10, CIFAR100, and Tiny ImageNet using the simple skip connection recurrent network and the Adam optimizer. As illustrated in <xref ref-type="fig" rid="F5">Figure 5</xref>, the BiPC model demonstrates effectiveness by mitigating the loss explosion on CIFAR10/100 and surpassing HPC performance on more complex datasets such as Tiny ImageNet.</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>The gradient norms of HPC and BiPC were evaluated across varying network depths and connection structures during the inference phase. The average gradient norm was determined for the 2<italic>th</italic> (shallow), 4<italic>th</italic> (middle), and 6<italic>th</italic> (deep) nodes, with the 6<italic>th</italic> node not incorporating skip connections. The <bold>Top subplot</bold>: the gradient norms for the HPC model. The <bold>Middle subplot</bold>: the BiPC gradients derived from the energy formulation expressed in <xref ref-type="disp-formula" rid="E10">Equation 10</xref>. The <bold>Bottom subplot</bold>: the gradient norms of the BiPC model based on the energy formulation presented in <xref ref-type="disp-formula" rid="E9">Equation 9</xref>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1605706-g0004.tif">
<alt-text>Graphs depicting gradient norms across different depths (Shallow, Middle, Deep) for HPC and BiPC gradient norms. Each set includes three line colors: red (feedforward), green (feedforward &#x0002B; recurrent), and blue (feedforward &#x0002B; recurrent &#x0002B; skip). Graphs show variations over epochs with distinct patterns in each configuration. </alt-text>
</graphic>
</fig>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Training loss comparison between the HPC and BiPC models across four benchmark datasets: <bold>(a)</bold> MNIST, <bold>(b)</bold> CIFAR10, <bold>(c)</bold> CIFAR100, and <bold>(d)</bold> Tiny ImageNet. Both HPC and BiPC are trained using the simple skip connection recurrent network (see <xref ref-type="table" rid="T1">Table 1</xref>) with the Adam optimizer. In all subfigures, the blue curve represents the HPC model, while the orange curve represents the BiPC model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1605706-g0005.tif">
<alt-text>Four line graphs compare the loss values of HPC and BiPC models across different datasets over epochs. (a) MNIST: Both models show decreasing trends, with BiPC slightly better. (b) CIFAR10: Both decline, BiPC&#x00027;s loss consistently lower. (c) CIFAR100: BiPC shows lower loss with variability, HPC has fluctuations. (d) TINY IMAGENET: BiPC consistently outperforms with lower loss than HPC. </alt-text>
</graphic>
</fig></sec>
<sec>
<title>6.3 Evaluation of the effectiveness of LALR</title>
<p>To evaluate the effectiveness of the LALR method, we analyzed weight gradient fluctuations and accuracy for BiPC/EP across three adaptive optimization algorithms: Adam, LALR, and LARS (<xref ref-type="bibr" rid="B58">You et al., 2017</xref>). <xref ref-type="fig" rid="F6">Figures 6a</xref>&#x02013;<xref ref-type="fig" rid="F6">c</xref> demonstrate the variations in weight gradients for BiPC when employing these optimizers within the 4<italic>th</italic> block of the generalized skip connection recurrent architecture. <xref ref-type="fig" rid="F6">Figures 6d</xref>&#x02013;<xref ref-type="fig" rid="F6">f</xref> depict the weight gradient variations for EP with different optimizers in the first layer of the hierarchical feedforward network. LALR ensures smoother, more stable gradient transitions and achieves stability more rapidly than LARS. <xref ref-type="table" rid="T3">Table 3</xref> presents the Top-1 validation accuracies of BiPC and LALR, respectively, compared against HPC, EP and backprop baselines across various datasets. The results indicate that LALR substantially enhances the accuracy of HPC, BiPC, and EP. Notably, BiPC combined with LALR achieves accuracy levels comparable to backprop within the same network architecture. However, we also observe lower performance on CIFAR-100 and Tiny ImageNet compared to MNIST and CIFAR-10. This discrepancy can be attributed to several factors. First, CIFAR-100 and Tiny ImageNet contain 100 and 200 categories respectively, with significant intra-class variability in object appearance, posture, and background. In contrast, MNIST and CIFAR-10 have only 10 well-separated categories with simpler visual patterns. The higher data complexity in CIFAR-100 and Tiny ImageNet increases the learning difficulty under fixed model capacity. Second, complex datasets often require deeper or wider networks with stronger feature representation capabilities to capture fine-grained distinctions between classes. As shown in <xref ref-type="table" rid="T3">Table 3</xref>, the skip-connection recurrent architecture, which contains more parameters and a more expressive structure than the hierarchical feedforward model, consistently outperforms the latter across all datasets. Third, our model inherits the Gaussian distribution assumption from the free-energy-based HPC framework. While this simplifies energy formulation and allows tractable optimization, it limits expressiveness when applied to high-resolution or highly non-Gaussian data. In real-world datasets such as CIFAR-100 and Tiny ImageNet, the pixel distributions are multimodal and deviate significantly from Gaussianity, which may lead to increased variational error and degraded performance.</p>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>Comparison of weight gradient fluctuations across various optimizers on the CIFAR10 dataset. Changes in weight gradients for the BiPC model when utilizing <bold>(a)</bold> Adam, <bold>(b)</bold> LARS, and <bold>(c)</bold> LALR optimizers. Similar plots for EP corresponding to a <bold>(d)</bold> EP Adam, <bold>(e)</bold> EP LARS, and <bold>(f)</bold> EP LALR optimizers.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1605706-g0006.tif">
<alt-text>There are six line graphs showing mean gradient or weight across epochs. Graphs (a), (b), and (c) depict BiPC methods with Adam, LARS, and LALR optimizers showing a reducing trend. Graphs (d), (e), and (f) depict EP methods with the same optimizers showing more variable trends. Each graph has labeled axes and a legend. </alt-text>
</graphic>
</fig>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Comparison of validation accuracies (%) for various approaches on the hierarchical feedforward and the skip connection recurrent networks.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Network</bold></th>
<th valign="top" align="center"><bold>Methods</bold></th>
<th valign="top" align="center"><bold>MNIST</bold></th>
<th valign="top" align="center"><bold>CIFAR10</bold></th>
<th valign="top" align="center"><bold>CIFAR100</bold></th>
<th valign="top" align="center"><bold>Tiny ImageNet</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Hierarchical feedforward network</td>
<td valign="top" align="center">HPC (Adam)</td>
<td valign="top" align="center">96.91 &#x000B1; 0.30</td>
<td valign="top" align="center">60.96 &#x000B1; 6.53</td>
<td valign="top" align="center">33.59 &#x000B1; 4.30</td>
<td valign="top" align="center">21.06 &#x000B1; 5.25</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">BiPC (Adam)</td>
<td valign="top" align="center">98.46 &#x000B1; 0.00</td>
<td valign="top" align="center">80.39 &#x000B1; 0.70</td>
<td valign="top" align="center">49.51 &#x000B1; 0.89</td>
<td valign="top" align="center">35.98 &#x000B1; 0.63</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">EP (Adam)</td>
<td valign="top" align="center">96.42 &#x000B1; 0.40</td>
<td valign="top" align="center">75.17 &#x000B1; 0.73</td>
<td valign="top" align="center">44.14 &#x000B1; 0.65</td>
<td valign="top" align="center">30.49 &#x000B1; 0.89</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">backprop (Adam)</td>
<td valign="top" align="center">97.63 &#x000B1; 0.00</td>
<td valign="top" align="center">81.69 &#x000B1; 0.01</td>
<td valign="top" align="center">52.47 &#x000B1; 0.01</td>
<td valign="top" align="center">35.70 &#x000B1; 0.00</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">HPC (LALR)</td>
<td valign="top" align="center">97.98 &#x000B1; 0.02</td>
<td valign="top" align="center">70.61 &#x000B1; 2.25</td>
<td valign="top" align="center">36.20 &#x000B1; 2.40</td>
<td valign="top" align="center">24.82 &#x000B1; 1.96</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">BiPC (LALR)</td>
<td valign="top" align="center">98.82 &#x000B1; 0.00</td>
<td valign="top" align="center">83.95 &#x000B1; 0.36</td>
<td valign="top" align="center">53.12 &#x000B1; 0.55</td>
<td valign="top" align="center">37.85 &#x000B1; 0.49</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">EP (LALR)</td>
<td valign="top" align="center">98.60 &#x000B1; 0.06</td>
<td valign="top" align="center">81.56 &#x000B1; 0.54</td>
<td valign="top" align="center">54.52 &#x000B1; 0.56</td>
<td valign="top" align="center">35.22 &#x000B1; 0.60</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">backprop (LALR)</td>
<td valign="top" align="center">98.95 &#x000B1; 0.00</td>
<td valign="top" align="center">82.91 &#x000B1; 0.00</td>
<td valign="top" align="center">54.01 &#x000B1; 0.02</td>
<td valign="top" align="center">36.14 &#x000B1; 0.01</td>
</tr> <tr>
<td valign="top" align="left">Skip connection recurrent network (simple)</td>
<td valign="top" align="center">HPC (Adam)</td>
<td valign="top" align="center">95.63 &#x000B1; 0.36</td>
<td valign="top" align="center">62.21 &#x000B1; 5.13</td>
<td valign="top" align="center">33.94 &#x000B1; 5.62</td>
<td valign="top" align="center">23.05 &#x000B1; 4.30</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">BiPC (Adam)</td>
<td valign="top" align="center">95.03 &#x000B1; 0.26</td>
<td valign="top" align="center">87.30 &#x000B1; 0.75</td>
<td valign="top" align="center">76.58 &#x000B1; 0.60</td>
<td valign="top" align="center">67.06 &#x000B1; 0.63</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">backprop (Adam)</td>
<td valign="top" align="center">98.84 &#x000B1; 0.05</td>
<td valign="top" align="center">90.05 &#x000B1; 0.07</td>
<td valign="top" align="center">77.63 &#x000B1; 0.10</td>
<td valign="top" align="center">70.83 &#x000B1; 0.26</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">HPC (LALR)</td>
<td valign="top" align="center">97.66 &#x000B1; 0.01</td>
<td valign="top" align="center">64.02 &#x000B1; 3.53</td>
<td valign="top" align="center">34.98 &#x000B1; 3.50</td>
<td valign="top" align="center">25.16 &#x000B1; 3.69</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">BiPC (LALR)</td>
<td valign="top" align="center">99.22 &#x000B1; 0.01</td>
<td valign="top" align="center">90.82 &#x000B1; 0.48</td>
<td valign="top" align="center">81.70 &#x000B1; 0.47</td>
<td valign="top" align="center">72.39 &#x000B1; 0.53</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">backprop (LALR)</td>
<td valign="top" align="center">98.89 &#x000B1; 0.02</td>
<td valign="top" align="center">91.46 &#x000B1; 0.42</td>
<td valign="top" align="center">82.40 &#x000B1; 0.55</td>
<td valign="top" align="center">71.98 &#x000B1; 0.55</td>
</tr> <tr>
<td valign="top" align="left">Skip connection recurrent network (complex)</td>
<td valign="top" align="center">HPC (Adam)</td>
<td valign="top" align="center">96.03 &#x000B1; 0.03</td>
<td valign="top" align="center">63.33 &#x000B1; 5.03</td>
<td valign="top" align="center">34.20 &#x000B1; 4.68</td>
<td valign="top" align="center">24.82 &#x000B1; 4.93</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">BiPC (Adam)</td>
<td valign="top" align="center">97.43 &#x000B1; 0.00</td>
<td valign="top" align="center">90.63 &#x000B1; 0.42</td>
<td valign="top" align="center">79.87 &#x000B1; 0.47</td>
<td valign="top" align="center">70.25 &#x000B1; 0.50</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">backprop (Adam)</td>
<td valign="top" align="center">98.51 &#x000B1; 0.00</td>
<td valign="top" align="center">94.25 &#x000B1; 0.43</td>
<td valign="top" align="center">80.65 &#x000B1; 0.45</td>
<td valign="top" align="center">70.58 &#x000B1; 0.49</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">HPC (LALR)</td>
<td valign="top" align="center">98.65 &#x000B1; 0.01</td>
<td valign="top" align="center">66.13 &#x000B1; 2.99</td>
<td valign="top" align="center">35.80 &#x000B1; 3.49</td>
<td valign="top" align="center">26.61 &#x000B1; 4.20</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">BiPC (LALR)</td>
<td valign="top" align="center">99.22 &#x000B1; 0.00</td>
<td valign="top" align="center">93.78 &#x000B1; 0.40</td>
<td valign="top" align="center">83.96 &#x000B1; 0.46</td>
<td valign="top" align="center">73.35 &#x000B1; 0.45</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">backprop (LALR)</td>
<td valign="top" align="center">98.89 &#x000B1; 0.01</td>
<td valign="top" align="center">94.56 &#x000B1; 0.46</td>
<td valign="top" align="center">81.60 &#x000B1; 0.49</td>
<td valign="top" align="center">74.18 &#x000B1; 0.50</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate the best accuracy results.</p>
</table-wrap-foot>
</table-wrap></sec>
<sec>
<title>6.4 Evaluation of the effectiveness of energy-based framework</title>
<p>To assess the reliability and efficiency of our framework, we trained the specified network using identical methods and hyperparameters on both PyTorch and our framework, utilizing a single NVIDIA A100 GPU. As shown in <xref ref-type="fig" rid="F7">Figure 7</xref>, our framework achieves comparable training accuracy and loss to PyTorch under identical settings, while reducing runtime by 50 percent.</p>
<fig position="float" id="F7">
<label>Figure 7</label>
<caption><p>Comparison of training accuracy, loss, and runtime between the our energy-based model with JAX implementation and the same model with PyTorch implementation on the CIFAR10 dataset. <bold>(a)</bold> Training curves of accuracy and loss using a batch size of 128 and input shape [32,32,3]. Solid curves denote training accuracy, while dashed curves denote training loss. Blue solid curve and red dashed curve correspond to our BiPC model with standard PyTorch implementation, whereas the green solid curve and orange dashed curve correspond to our BiPC model with JAX implementation. <bold>(b)</bold> Runtime comparison per epoch. The red bars indicate the training time of our BiPC model with JAX implementation, while the blue bars show the PyTorch baseline. Across all epochs, our method significantly reduces computational time, demonstrating better efficiency.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1605706-g0007.tif">
<alt-text>(a) Line chart comparing training accuracy and loss for PyTorch and a proposed method on CIFAR10 with batch size 128. PyTorch accuracy and loss converge faster than the proposed method. (b) Bar chart showing training time in seconds. The proposed method (red) takes roughly half the time of PyTorch (blue) for each epoch.</alt-text>
</graphic>
</fig>
</sec></sec>
<sec id="s7">
<title>7 Discussion and conclusion</title>
<p>This study aims to address gradient explosion and vanishing issues in EBLL models, such as classic hierarchical PC, during ANN training. To enhance training efficiency, we developed a JAX-based framework. Drawing on neuroscience and AI engineering, we introduce a novel BiPC model based on a biologically inspired ANN. BiPC utilizes energy from forward and backward processes to constrain updates and prevent gradient explosion during local state and parameter optimization. Experiments reveal that while bidirectional energy constraints effectively address gradient explosion in deep ANNs, resolving gradient vanishing necessitates incorporating skip connections. BiPC with recurrent and skip connections surpasses models relying solely on feedforward and feedback connections. To address gradient variability across layers during local updates, we propose the LALR method, optimizing gradient descent for state and parameter updates in energy-based PC and EP. This method significantly improves target recognition performance. BiPC with LALR achieves object recognition accuracy comparable to backprop; however, LALR alone cannot fully address gradient explosion in hierarchical PC. Our EBLL framework, structured with points and edges, demonstrates superior accuracy and faster execution compared to PyTorch in hierarchical and recurrent architectures.</p>
<p>Our model and framework have some limitations. While BiPC resolves gradient explosion, it does not fully address gradient vanishing, likely due to both model directions producing low energy, leading to gradient decay. Additionally, BiPC focuses on a single cortical region without accounting for inter-unit interactions across the entire brain. Furthermore, this model is trained solely on biologically inspired neural networks, excluding biomimetic networks such as spiking neural networks (SNNs) or brain emulation networks. This study focuses on optimizing EBLL algorithms for practical applications in ANNs, rather than theoretical study. Future work will investigate energy-based models in biomimetic networks, such as SNNs and brain emulation systems. Additionally, the bidirectional concept in the BiPC method has yet to be effectively applied to local learning approaches based on Hopfield energy, such as EP. This trend is because Hopfield energy originates from a fully connected network, making its energy undirected. Additionally, the EBLL framework uses a graph structure, theoretically enabling its application to networks with arbitrary topologies, including large-scale brain simulation networks with millions of nodes and edges. However, the framework lacks effective memory management, limiting its scalability for large networks. Improving efficient learning support for large-scale networks is a critical challenge. Currently, brain simulation networks (<xref ref-type="bibr" rid="B43">Potjans and Diesmann, 2014</xref>) have not successfully handled complex tasks such as vision and text. Therefore, overcoming these limitations and advancing EBLL methods for intelligent task mastery in large-scale brain simulations is a key research objective.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s8">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>, further inquiries can be directed to the corresponding author.</p></sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>HC: Conceptualization, Supervision, Methodology, Formal analysis, Software, Data curation, Writing &#x02013; review &#x00026; editing, Writing &#x02013; original draft, Project administration. BY: Methodology, Writing &#x02013; review &#x00026; editing, Investigation, Project administration, Formal analysis, Writing &#x02013; original draft, Resources, Data curation. FH: Methodology, Software, Writing &#x02013; review &#x00026; editing, Investigation, Resources. FZ: Resources, Visualization, Validation, Writing &#x02013; original draft, Conceptualization, Methodology. SC: Data curation, Resources, Project administration, Formal analysis, Writing &#x02013; original draft. CW: Investigation, Conceptualization, Project administration, Writing &#x02013; review &#x00026; editing, Formal analysis, Methodology, Data curation. FL: Visualization, Methodology, Formal analysis, Conceptualization, Software, Writing &#x02013; review &#x00026; editing, Investigation. YC: Formal analysis, Funding acquisition, Project administration, Writing &#x02013; original draft, Data curation, Methodology, Conceptualization, Software.</p>
</sec>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. The authors gratefully acknowledge the support from the project &#x0201C;Key Technologies for Heterogeneous and Brain-Inspired Computing in Power Systems&#x0201D; [Grant No. 5700-202358838A-4-3-WL].</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>FZ, SC, CW, and FL were employed by State Grid Shanghai Municipal Electric Power Company. The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest. The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p></sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec><sec sec-type="supplementary-material" id="s13">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2025.1605706/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frai.2025.1605706/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Supplementary_file_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Amit</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>Deep learning with asymmetric connections and hebbian updates</article-title>. <source>Front. Comput. Neurosci</source>. <volume>13</volume>:<fpage>18</fpage>. <pub-id pub-id-type="doi">10.3389/fncom.2019.00018</pub-id><pub-id pub-id-type="pmid">31019458</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Angelucci</surname> <given-names>A.</given-names></name> <name><surname>Petreanu</surname> <given-names>L.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Feedforward and feedback connections: functional connectivity, synaptic physiology, and function,&#x0201D;</article-title> in <source>The Cerebral Cortex and Thalamus</source>, 405&#x02013;418. <pub-id pub-id-type="doi">10.1093/med/9780197676158.003.0038</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>2014</year>). <article-title>How auto-encoders could provide credit assignment in deep networks via target propagation</article-title>. <source>arXiv</source> [Preprint]. arXiv:1407.7906. <pub-id pub-id-type="doi">10.48550/arXiv.1407.7906</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Fischer</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>Early inference in energy-based models approximates back-propagation</article-title>. <source>arXiv</source> [Preprint]. arXiv:1510.02777. <pub-id pub-id-type="doi">10.48550/arXiv.1510.02777</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boccato</surname> <given-names>T.</given-names></name> <name><surname>Ferrante</surname> <given-names>M.</given-names></name> <name><surname>Duggento</surname> <given-names>A.</given-names></name> <name><surname>Toschi</surname> <given-names>N.</given-names></name></person-group> (<year>2024</year>). <article-title>Beyond multilayer perceptrons: Investigating complex topologies in neural networks</article-title>. <source>Neural Netw</source>. <volume>171</volume>, <fpage>215</fpage>&#x02013;<lpage>228</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2023.12.012</pub-id><pub-id pub-id-type="pmid">38096650</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Bradbury</surname> <given-names>J.</given-names></name> <name><surname>Frostig</surname> <given-names>R.</given-names></name> <name><surname>Hawkins</surname> <given-names>P.</given-names></name> <name><surname>Johnson</surname> <given-names>M. J.</given-names></name> <name><surname>Leary</surname> <given-names>C.</given-names></name> <name><surname>Maclaurin</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2018</year>). <source>JAX: Composable Transformations of Python</source>&#x0002B;<italic>NumPy Programs</italic>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://github.com/google/jax">http://github.com/google/jax</ext-link></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Buckley</surname> <given-names>C. L.</given-names></name> <name><surname>Kim</surname> <given-names>C. S.</given-names></name> <name><surname>McGregor</surname> <given-names>S.</given-names></name> <name><surname>Seth</surname> <given-names>A. K.</given-names></name></person-group> (<year>2017</year>). <article-title>The free energy principle for action and perception: a mathematical review</article-title>. <source>J. Math. Psychol</source>. <volume>81</volume>, <fpage>55</fpage>&#x02013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmp.2017.09.004</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>A.</given-names></name> <name><surname>Ping</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Xiao</surname> <given-names>X.</given-names></name> <name><surname>Yin</surname> <given-names>C.</given-names></name> <name><surname>Nazarian</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>&#x0201C;Unlocking deep learning: a bp-free approach for parallel block-wise training of neural networks,&#x0201D;</article-title> in <source>ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP</source>) (<publisher-loc>Seoul</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4235</fpage>&#x02013;<lpage>4239</lpage>. <pub-id pub-id-type="doi">10.1109/ICASSP48485.2024.10447377</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clark</surname> <given-names>A.</given-names></name></person-group> (<year>2013</year>). <article-title>Whatever next? Predictive brains, situated agents, and the future of cognitive science</article-title>. <source>Behav. Brain Sci</source>. <volume>36</volume>, <fpage>181</fpage>&#x02013;<lpage>204</lpage>. <pub-id pub-id-type="doi">10.1017/S0140525X12000477</pub-id><pub-id pub-id-type="pmid">23663408</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Crick</surname> <given-names>F.</given-names></name></person-group> (<year>1989</year>). <article-title>The recent excitement about neural networks</article-title>. <source>Nature</source> <volume>337</volume>, <fpage>129</fpage>&#x02013;<lpage>132</lpage>. <pub-id pub-id-type="doi">10.1038/337129a0</pub-id><pub-id pub-id-type="pmid">2911347</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dold</surname> <given-names>D.</given-names></name> <name><surname>Kungl</surname> <given-names>A. F.</given-names></name> <name><surname>Sacramento</surname> <given-names>J.</given-names></name> <name><surname>Petrovici</surname> <given-names>M. A.</given-names></name> <name><surname>Schindler</surname> <given-names>K.</given-names></name> <name><surname>Binas</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Lagrangian dynamics of dendritic microcircuits enables real-time backpropagation of errors</article-title>. <source>Target</source> <volume>100</volume>:<fpage>2</fpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duchi</surname> <given-names>J.</given-names></name> <name><surname>Hazan</surname> <given-names>E.</given-names></name> <name><surname>Singer</surname> <given-names>Y.</given-names></name></person-group> (<year>2011</year>). <article-title>Adaptive subgradient methods for online learning and stochastic optimization</article-title>. <source>J. Mach. Learn. Res.</source> <volume>12</volume>, <fpage>2121</fpage>&#x02013;<lpage>2159</lpage>. <pub-id pub-id-type="doi">10.5555/1953048.2021068</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friston</surname> <given-names>K.</given-names></name></person-group> (<year>2003</year>). <article-title>Learning and inference in the brain</article-title>. <source>Neural Netw</source>. <volume>16</volume>, <fpage>1325</fpage>&#x02013;<lpage>1352</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2003.06.005</pub-id><pub-id pub-id-type="pmid">14622888</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friston</surname> <given-names>K.</given-names></name></person-group> (<year>2005</year>). <article-title>A theory of cortical responses</article-title>. <source>Philos. Trans. R. Soc. B: Biol. Sci</source>. <volume>360</volume>, <fpage>815</fpage>&#x02013;<lpage>836</lpage>. <pub-id pub-id-type="doi">10.1098/rstb.2005.1622</pub-id><pub-id pub-id-type="pmid">15937014</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friston</surname> <given-names>K.</given-names></name></person-group> (<year>2010</year>). <article-title>The free-energy principle: a unified brain theory?</article-title> <source>Nat. Rev. Neurosci</source>. <volume>11</volume>, <fpage>127</fpage>&#x02013;<lpage>138</lpage>. <pub-id pub-id-type="doi">10.1038/nrn2787</pub-id><pub-id pub-id-type="pmid">20068583</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friston</surname> <given-names>K.</given-names></name> <name><surname>Kiebel</surname> <given-names>S.</given-names></name></person-group> (<year>2009</year>). <article-title>Predictive coding under the free-energy principle</article-title>. <source>Philos. Trans. R. Soc. B: Biol. Sci</source>. <volume>364</volume>, <fpage>1211</fpage>&#x02013;<lpage>1221</lpage>. <pub-id pub-id-type="doi">10.1098/rstb.2008.0300</pub-id><pub-id pub-id-type="pmid">19528002</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friston</surname> <given-names>K.</given-names></name> <name><surname>Kilner</surname> <given-names>J.</given-names></name> <name><surname>Harrison</surname> <given-names>L.</given-names></name></person-group> (<year>2006</year>). <article-title>A free energy principle for the brain</article-title>. <source>J. Physiol</source>. <volume>100</volume>, <fpage>70</fpage>&#x02013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1016/j.jphysparis.2006.10.001</pub-id><pub-id pub-id-type="pmid">17097864</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friston</surname> <given-names>K. J.</given-names></name> <name><surname>Stephan</surname> <given-names>K. E.</given-names></name></person-group> (<year>2007</year>). <article-title>Free-energy and the brain</article-title>. <source>Synthese</source> <volume>159</volume>, <fpage>417</fpage>&#x02013;<lpage>458</lpage>. <pub-id pub-id-type="doi">10.1007/s11229-007-9237-y</pub-id><pub-id pub-id-type="pmid">19325932</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Frostig</surname> <given-names>R.</given-names></name> <name><surname>Johnson</surname> <given-names>M. J.</given-names></name> <name><surname>Leary</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>Compiling machine learning programs via high-level tracing</article-title>. <source>Syst. Mach. Learn</source>. <volume>4</volume>, <fpage>1</fpage>&#x02013;<lpage>3</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2003</year>). <article-title>The ups and downs of hebb synapses</article-title>. <source>Can. Psychol./Psychol. Canad</source>. <volume>44</volume>:<fpage>10</fpage>. <pub-id pub-id-type="doi">10.1037/h0085812</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2022</year>). <article-title>The forward-forward algorithm: some preliminary investigations</article-title>. <source>arXiv</source> [Preprint]. arXiv:2212.13345. <pub-id pub-id-type="doi">10.48550/arXiv.2212.13345</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hirani</surname> <given-names>G.</given-names></name> <name><surname>Kevin</surname> <given-names>I.</given-names></name> <name><surname>Wang</surname> <given-names>K.</given-names></name> <name><surname>Abdulla</surname> <given-names>W.</given-names></name></person-group> (<year>2024</year>). <article-title>A scalable unsupervised and back propagation free learning with sacsom: a novel approach to SOM-based architectures</article-title>. <source>IEEE Trans. Artif. Intell</source>. <volume>6</volume>, <fpage>955</fpage>&#x02013;<lpage>967</lpage>. <pub-id pub-id-type="doi">10.1109/TAI.2024.3504479</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hoffmann</surname> <given-names>C.</given-names></name> <name><surname>Cho</surname> <given-names>E.</given-names></name> <name><surname>Zalesky</surname> <given-names>A.</given-names></name> <name><surname>Di Biase</surname> <given-names>M. A.</given-names></name></person-group> (<year>2024</year>). <article-title>From pixels to connections: exploring <italic>in vitro</italic> neuron reconstruction software for network graph generation</article-title>. <source>Commun. Biol</source>. <volume>7</volume>:<fpage>571</fpage>. <pub-id pub-id-type="doi">10.1038/s42003-024-06264-9</pub-id><pub-id pub-id-type="pmid">38750282</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hopfield</surname> <given-names>J. J.</given-names></name></person-group> (<year>1982</year>). <article-title>Neural networks and physical systems with emergent collective computational abilities</article-title>. <source>Proc. Nat. Acad. Sci</source>. <volume>79</volume>, <fpage>2554</fpage>&#x02013;<lpage>2558</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.79.8.2554</pub-id><pub-id pub-id-type="pmid">6953413</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Khacef</surname> <given-names>L.</given-names></name> <name><surname>Miramond</surname> <given-names>B.</given-names></name> <name><surname>Barrientos</surname> <given-names>D.</given-names></name> <name><surname>Upegui</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Self-organizing neurons: toward brain-inspired unsupervised learning,&#x0201D;</article-title> in <source>2019 International Joint Conference on Neural Networks (IJCNN)</source> (<publisher-loc>Budapest</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/IJCNN.2019.8852098</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kinghorn</surname> <given-names>P. F.</given-names></name> <name><surname>Millidge</surname> <given-names>B.</given-names></name> <name><surname>Buckley</surname> <given-names>C. L.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Preventing deterioration of classification accuracy in predictive coding networks,&#x0201D;</article-title> in <source>International Workshop on Active Inference</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-28719-0_1</pub-id><pub-id pub-id-type="pmid">38553167</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D. P.</given-names></name> <name><surname>Ba</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). <article-title>Adam: a method for stochastic optimization</article-title>. <source>arXiv</source> [Preprint]. arXiv:1412.6980. <pub-id pub-id-type="doi">10.48550/arXiv.1412.6980</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krotov</surname> <given-names>D.</given-names></name> <name><surname>Hopfield</surname> <given-names>J. J.</given-names></name></person-group> (<year>2019</year>). <article-title>Unsupervised learning by competing hidden units</article-title>. <source>Proc. Nat. Acad. Sci</source>., <volume>116</volume>, <fpage>7723</fpage>&#x02013;<lpage>7731</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1820458116</pub-id><pub-id pub-id-type="pmid">30926658</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kubilius</surname> <given-names>J.</given-names></name> <name><surname>Schrimpf</surname> <given-names>M.</given-names></name> <name><surname>Nayebi</surname> <given-names>A.</given-names></name> <name><surname>Bear</surname> <given-names>D.</given-names></name> <name><surname>Yamins</surname> <given-names>D. L.</given-names></name> <name><surname>DiCarlo</surname> <given-names>J. J.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Cornet: modeling the neural mechanisms of core object recognition</article-title>. <source>bioRxiv</source>. <pub-id pub-id-type="doi">10.1101/408385</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>D.-H.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Fischer</surname> <given-names>A.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Difference target propagation,&#x0201D;</article-title> in <source>Machine Learning and Knowledge Discovery in Databases: European Conference, ECML PKDD 2015, Porto, Portugal, September 7-11, 2015, Proceedings, Part I 15</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>498</fpage>&#x02013;<lpage>515</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-23528-8_31</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lillicrap</surname> <given-names>T. P.</given-names></name> <name><surname>Cownden</surname> <given-names>D.</given-names></name> <name><surname>Tweed</surname> <given-names>D. B.</given-names></name> <name><surname>Akerman</surname> <given-names>C. J.</given-names></name></person-group> (<year>2016</year>). <article-title>Random synaptic feedback weights support error backpropagation for deep learning</article-title>. <source>Nat. Commun</source>. <volume>7</volume>:<fpage>13276</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms13276</pub-id><pub-id pub-id-type="pmid">27824044</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lillicrap</surname> <given-names>T. P.</given-names></name> <name><surname>Santoro</surname> <given-names>A.</given-names></name> <name><surname>Marris</surname> <given-names>L.</given-names></name> <name><surname>Akerman</surname> <given-names>C. J.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2020</year>). <article-title>Backpropagation and the brain</article-title>. <source>Nat. Rev. Neurosci</source>. <volume>21</volume>, <fpage>335</fpage>&#x02013;<lpage>346</lpage>. <pub-id pub-id-type="doi">10.1038/s41583-020-0277-3</pub-id><pub-id pub-id-type="pmid">32303713</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>H.</given-names></name> <name><surname>Fu</surname> <given-names>J.</given-names></name> <name><surname>Glass</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>Adaptive bidirectional backpropagation: towards biologically plausible error signal transmission in neural networks</article-title>. <source>arXiv</source> [Preprint]. arXiv:1702.07097. <pub-id pub-id-type="doi">10.48550/arXiv.1702.07097</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Millidge</surname> <given-names>B.</given-names></name> <name><surname>Seth</surname> <given-names>A.</given-names></name> <name><surname>Buckley</surname> <given-names>C. L.</given-names></name></person-group> (<year>2021</year>). <article-title>Predictive coding: a theoretical and experimental review</article-title>. <source>arXiv</source> [Preprint]. arXiv:2107.12979. <pub-id pub-id-type="doi">10.48550/arXiv.2107.12979</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Millidge</surname> <given-names>B.</given-names></name> <name><surname>Song</surname> <given-names>Y.</given-names></name> <name><surname>Salvatori</surname> <given-names>T.</given-names></name> <name><surname>Lukasiewicz</surname> <given-names>T.</given-names></name> <name><surname>Bogacz</surname> <given-names>R.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Backpropagation at the infinitesimal inference limit of energy-based models: unifying predictive coding, equilibrium propagation, and contrastive Hebbian learning,&#x0201D;</article-title> in <source>The Eleventh International Conference on Learning Representations</source> (<publisher-loc>Kigali</publisher-loc>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=nIMifqu2EO">https://openreview.net/forum?id=nIMifqu2EO</ext-link></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Millidge</surname> <given-names>B.</given-names></name> <name><surname>Tschantz</surname> <given-names>A.</given-names></name> <name><surname>Buckley</surname> <given-names>C. L.</given-names></name></person-group> (<year>2022</year>). <article-title>Predictive coding approximates backprop along arbitrary computation graphs</article-title>. <source>Neural Comput</source>. <volume>34</volume>, <fpage>1329</fpage>&#x02013;<lpage>1368</lpage>. <pub-id pub-id-type="doi">10.1162/neco_a_01497</pub-id><pub-id pub-id-type="pmid">35534010</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Momeni</surname> <given-names>A.</given-names></name> <name><surname>Rahmani</surname> <given-names>B. Mall&#x000E9;jac, M.</given-names></name> <name><surname>Del Hougne</surname> <given-names>P.</given-names></name> <name><surname>Fleury</surname> <given-names>R.</given-names></name></person-group> (<year>2023</year>). <article-title>Backpropagation-free training of deep physical neural networks</article-title>. <source>Science</source> <volume>382</volume>, <fpage>1297</fpage>&#x02013;<lpage>1303</lpage>. <pub-id pub-id-type="doi">10.1126/science.adi8474</pub-id><pub-id pub-id-type="pmid">37995209</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moraitis</surname> <given-names>T.</given-names></name> <name><surname>Toichkin</surname> <given-names>D. Journ&#x000E9;, A.</given-names></name> <name><surname>Chua</surname> <given-names>Y.</given-names></name> <name><surname>Guo</surname> <given-names>Q.</given-names></name></person-group> (<year>2022</year>). <article-title>Softhebb: Bayesian inference in unsupervised hebbian soft winner-take-all networks</article-title>. <source>Neuromorphic Comput. Eng</source>. <volume>2</volume>:<fpage>044017</fpage>. <pub-id pub-id-type="doi">10.1088/2634-4386/aca710</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>N&#x000F8;kland</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Direct feedback alignment provides learning in deep neural networks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds. D. Lee, M. Sugiyama, U. Luxburg, I. Guyon, and R. Garnett (Barcelona: Curran Associates, Inc.), Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2016/file/d490d7b4576290fa60eb31b5fc917ad1-Paper.pdf">https://proceedings.neurips.cc/paper_files/paper/2016/file/d490d7b4576290fa60eb31b5fc917ad1-Paper.pdf</ext-link></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Piekarski</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Incorporating (variational) free energy models into mechanisms: the case of predictive processing under the free energy principle</article-title>. <source>Synthese</source> <volume>202</volume>:<fpage>58</fpage>. <pub-id pub-id-type="doi">10.1007/s11229-023-04292-2</pub-id></citation>
</ref>
<ref id="B41">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Pinchetti</surname> <given-names>L.</given-names></name> <name><surname>Salvatori</surname> <given-names>T.</given-names></name> <name><surname>Yordanov</surname> <given-names>Y.</given-names></name> <name><surname>Millidge</surname> <given-names>B.</given-names></name> <name><surname>Song</surname> <given-names>Y.</given-names></name> <name><surname>Lukasiewicz</surname> <given-names>T.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Predictive coding beyond Gaussian distributions&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds. A. H. Oh, A. Agarwal, D. Belgrave, and K. Cho (New Orleans, LA). Available online at: <ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=Ryy7tVvBUk">https://openreview.net/forum?id=Ryy7tVvBUk</ext-link></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pogodin</surname> <given-names>R.</given-names></name> <name><surname>Latham</surname> <given-names>P.</given-names></name></person-group> (<year>2020</year>). <article-title>Kernelized information bottleneck leads to biologically plausible 3-factor hebbian learning in deep networks</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>33</volume>, <fpage>7296</fpage>&#x02013;<lpage>7307</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Potjans</surname> <given-names>T. C.</given-names></name> <name><surname>Diesmann</surname> <given-names>M.</given-names></name></person-group> (<year>2014</year>). <article-title>The cell-type specific cortical microcircuit: relating structure and activity in a full-scale spiking network model</article-title>. <source>Cereb. Cortex</source> <volume>24</volume>, <fpage>785</fpage>&#x02013;<lpage>806</lpage>. <pub-id pub-id-type="doi">10.1093/cercor/bhs358</pub-id><pub-id pub-id-type="pmid">23203991</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rao</surname> <given-names>R. P.</given-names></name> <name><surname>Ballard</surname> <given-names>D. H.</given-names></name></person-group> (<year>1999</year>). <article-title>Predictive coding in the visual cortex: a functional interpretation of some extra-classical receptive-field effects</article-title>. <source>Nat. Neurosci</source>. <volume>2</volume>, <fpage>79</fpage>&#x02013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1038/4580</pub-id><pub-id pub-id-type="pmid">10195184</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rockland</surname> <given-names>K. S.</given-names></name></person-group> (<year>2022</year>). <article-title>Notes on visual cortical feedback and feedforward connections</article-title>. <source>Front. Syst. Neurosci</source>. <volume>16</volume>:<fpage>784310</fpage>. <pub-id pub-id-type="doi">10.3389/fnsys.2022.784310</pub-id><pub-id pub-id-type="pmid">35153685</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rosenbaum</surname> <given-names>R.</given-names></name></person-group> (<year>2022</year>). <article-title>On the relationship between predictive coding and backpropagation</article-title>. <source>PLoS ONE</source> <volume>17</volume>:<fpage>e0266102</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0266102</pub-id><pub-id pub-id-type="pmid">35358258</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sa-Couto</surname> <given-names>L.</given-names></name> <name><surname>Wichert</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Self-organizing maps on &#x0201C;what-where&#x0201D; codes towards fully unsupervised classification</article-title>. <source>Biol. Cybern</source>. <volume>117</volume>, <fpage>211</fpage>&#x02013;<lpage>220</lpage>. <pub-id pub-id-type="doi">10.1007/s00422-023-00963-y</pub-id><pub-id pub-id-type="pmid">37188974</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Salova</surname> <given-names>A.</given-names></name> <name><surname>Kov&#x000E1;cs</surname> <given-names>I. A.</given-names></name></person-group> (<year>2025</year>). <article-title>Combined topological and spatial constraints are required to capture the structure of neural connectomes</article-title>. <source>Netw. Neurosci</source>. <volume>9</volume>, <fpage>181</fpage>&#x02013;<lpage>206</lpage>. <pub-id pub-id-type="doi">10.1162/netn_a_00428</pub-id><pub-id pub-id-type="pmid">40161988</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Scellier</surname> <given-names>B.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>2017</year>). <article-title>Equilibrium propagation: bridging the gap between energy-based models and backpropagation</article-title>. <source>Front. Comput. Neurosci</source>. <volume>11</volume>:<fpage>24</fpage>. <pub-id pub-id-type="doi">10.3389/fncom.2017.00024</pub-id><pub-id pub-id-type="pmid">28522969</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Spratling</surname> <given-names>M. W.</given-names></name></person-group> (<year>2017</year>). <article-title>A review of predictive coding algorithms</article-title>. <source>Brain Cogn</source>. <volume>112</volume>, <fpage>92</fpage>&#x02013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.1016/j.bandc.2015.11.003</pub-id><pub-id pub-id-type="pmid">26809759</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Stork</surname> <given-names>D. G.</given-names></name></person-group> (<year>1989</year>). <article-title>&#x0201C;Is backpropagation biologically plausible,&#x0201D;</article-title> in <source>International Joint Conference on Neural Networks, Volume 2</source> (<publisher-loc>Washington, DC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>241</fpage>&#x02013;<lpage>246</lpage>. <pub-id pub-id-type="doi">10.1109/IJCNN.1989.118705</pub-id></citation>
</ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Whittington</surname> <given-names>J. C.</given-names></name> <name><surname>Bogacz</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>An approximation of the error backpropagation algorithm in a predictive coding network with local hebbian synaptic plasticity</article-title>. <source>Neural Comput</source>. <volume>29</volume>, <fpage>1229</fpage>&#x02013;<lpage>1262</lpage>. <pub-id pub-id-type="doi">10.1162/NECO_a_00949</pub-id><pub-id pub-id-type="pmid">28333583</pub-id></citation></ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiao</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Bogdan</surname> <given-names>P.</given-names></name></person-group> (<year>2021</year>). <article-title>Deciphering the generating rules and functionalities of complex networks</article-title>. <source>Sci. Rep</source>. <volume>11</volume>:<fpage>22964</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-02203-4</pub-id><pub-id pub-id-type="pmid">34824290</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xie</surname> <given-names>X.</given-names></name> <name><surname>Seung</surname> <given-names>H. S.</given-names></name></person-group> (<year>2003</year>). <article-title>Equivalence of backpropagation and contrastive hebbian learning in a layered network</article-title>. <source>Neural Comput</source>. <volume>15</volume>, <fpage>441</fpage>&#x02013;<lpage>454</lpage>. <pub-id pub-id-type="doi">10.1162/089976603762552988</pub-id><pub-id pub-id-type="pmid">12590814</pub-id></citation></ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>R.</given-names></name> <name><surname>Sala</surname> <given-names>F.</given-names></name> <name><surname>Bogdan</surname> <given-names>P.</given-names></name></person-group> (<year>2021</year>). <article-title>Hidden network generating rules from partially observed complex networks</article-title>. <source>Commun. Phys</source>. <volume>4</volume>:<fpage>199</fpage>. <pub-id pub-id-type="doi">10.1038/s42005-021-00701-5</pub-id></citation>
</ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>C.</given-names></name> <name><surname>Cheng</surname> <given-names>M.</given-names></name> <name><surname>Xiao</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Nazarian</surname> <given-names>S.</given-names></name> <name><surname>Irimia</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Leader-follower neural networks with local error signals inspired by complex collectives</article-title>. <source>arXiv</source> [Preprint] arXiv:2310.07885. <pub-id pub-id-type="doi">10.48550/arXiv.2310.07885</pub-id></citation>
</ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>C.</given-names></name> <name><surname>Xiao</surname> <given-names>X.</given-names></name> <name><surname>Balaban</surname> <given-names>V.</given-names></name> <name><surname>Kandel</surname> <given-names>M. E.</given-names></name> <name><surname>Lee</surname> <given-names>Y. J.</given-names></name> <name><surname>Popescu</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Network science characteristics of brain-derived neuronal cultures deciphered from quantitative phase imaging data</article-title>. <source>Sci. Rep</source>. <volume>10</volume>:<fpage>15078</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-72013-7</pub-id><pub-id pub-id-type="pmid">32934305</pub-id></citation></ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>You</surname> <given-names>Y.</given-names></name> <name><surname>Gitman</surname> <given-names>I.</given-names></name> <name><surname>Ginsburg</surname> <given-names>B.</given-names></name></person-group> (<year>2017</year>). <article-title>Large batch training of convolutional networks</article-title>. <source>arXiv</source> [Preprint] arXiv:1708.03888. <pub-id pub-id-type="doi">10.48550/arXiv.1708.03888</pub-id></citation>
</ref>
<ref id="B59">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>You</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Reddi</surname> <given-names>S.</given-names></name> <name><surname>Hseu</surname> <given-names>J.</given-names></name> <name><surname>Kumar</surname> <given-names>S.</given-names></name> <name><surname>Bhojanapalli</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Large batch optimization for deep learning: training BERT in 76 minutes,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source> (<publisher-loc>Addis Ababa</publisher-loc>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=Syx4wnEtvH">https://openreview.net/forum?id=Syx4wnEtvH</ext-link></citation>
</ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Znaidi</surname> <given-names>M. R.</given-names></name> <name><surname>Sia</surname> <given-names>J.</given-names></name> <name><surname>Ronquist</surname> <given-names>S.</given-names></name> <name><surname>Rajapakse</surname> <given-names>I.</given-names></name> <name><surname>Jonckheere</surname> <given-names>E.</given-names></name> <name><surname>Bogdan</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>A unified approach of detecting phase transition in time-varying complex networks</article-title>. <source>Sci. Rep</source>. <volume>13</volume>:<fpage>17948</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-44791-3</pub-id><pub-id pub-id-type="pmid">37864007</pub-id></citation></ref>
</ref-list>
</back>
</article> 