<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2024.1360300</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Co-learning synaptic delays, weights and adaptation in spiking neural networks</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Deckers</surname> <given-names>Lucas</given-names></name>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1722189/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Van Damme</surname> <given-names>Laurens</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/2673054/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Van Leekwijck</surname> <given-names>Werner</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/2048012/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Tsang</surname> <given-names>Ing Jyh</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/1129420/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Latr&#x000E9;</surname> <given-names>Steven</given-names></name>
<uri xlink:href="http://loop.frontiersin.org/people/1436057/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff><institution>IDLab, imec, University of Antwerp</institution>, <addr-line>Antwerp</addr-line>, <country>Belgium</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Bernabe Linares-Barranco, Spanish National Research Council (CSIC), Spain</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Timoth&#x000E9;e Masquelier, Centre National de la Recherche Scientifique (CNRS), France</p>
<p>Pablo Negri, National Scientific and Technical Research Council (CONICET), Argentina</p>
<p>Sander Bohte, Centrum Wiskunde &#x00026; Informatica, Netherlands</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Lucas Deckers <email>lucas.deckers&#x00040;uantwerpen.be</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>12</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>18</volume>
<elocation-id>1360300</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>20</day>
<month>03</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Deckers, Van Damme, Van Leekwijck, Tsang and Latr&#x000E9;.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Deckers, Van Damme, Van Leekwijck, Tsang and Latr&#x000E9;</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license></permissions>
<abstract>
<p>Spiking neural network (SNN) distinguish themselves from artificial neural network (ANN) because of their inherent temporal processing and spike-based computations, enabling a power-efficient implementation in neuromorphic hardware. In this study, we demonstrate that data processing with spiking neurons can be enhanced by co-learning the synaptic weights with two other biologically inspired neuronal features: (1) a set of parameters describing neuronal adaptation processes and (2) synaptic propagation delays. The former allows a spiking neuron to learn how to specifically react to incoming spikes based on its past. The trained adaptation parameters result in neuronal heterogeneity, which leads to a greater variety in available spike patterns and is also found in the brain. The latter enables to learn to explicitly correlate spike trains that are temporally distanced. Synaptic delays reflect the time an action potential requires to travel from one neuron to another. We show that each of the co-learned features separately leads to an improvement over the baseline SNN and that the combination of both leads to state-of-the-art SNN results on all speech recognition datasets investigated with a simple 2-hidden layer feed-forward network. Our SNN outperforms the benchmark ANN on the neuromorphic datasets (Spiking Heidelberg Digits and Spiking Speech Commands), even with fewer trainable parameters. On the 35-class Google Speech Commands dataset, our SNN also outperforms a GRU of similar size. Our study presents brain-inspired improvements in SNN that enable them to excel over an equivalent ANN of similar size on tasks with rich temporal dynamics.</p></abstract>
<kwd-group>
<kwd>spiking neural networks</kwd>
<kwd>synaptic delays</kwd>
<kwd>neuronal adaptation</kwd>
<kwd>co-learning</kwd>
<kwd>speech recognition</kwd>
<kwd>surrogate gradients</kwd>
</kwd-group>
<contract-sponsor id="cn001">Fonds Wetenschappelijk Onderzoek<named-content content-type="fundref-id">10.13039/501100003130</named-content></contract-sponsor>
<counts>
<fig-count count="4"/>
<table-count count="3"/>
<equation-count count="9"/>
<ref-count count="50"/>
<page-count count="11"/>
<word-count count="7994"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Neuromorphic Engineering</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Spiking neural networks (SNN), seen as the third generation neural network models (Maass, <xref ref-type="bibr" rid="B23">1997</xref>), have recently attracted growing attention as a low-power alternative for artificial neural networks (ANN). Unlike ANN implementations (Garc&#x00301;&#x00131;a-Mart&#x00301;&#x00131;n et al., <xref ref-type="bibr" rid="B12">2019</xref>), SNN can enable power-efficient processing on neuromorphic hardware such as SENeCa (Yousefzadeh et al., <xref ref-type="bibr" rid="B46">2022</xref>), the Intel Loihi2 (Orchard et al., <xref ref-type="bibr" rid="B27">2021</xref>), or IBM TrueNorth (DeBole et al., <xref ref-type="bibr" rid="B6">2019</xref>). The main power gains can be attributed to SNN inherent event-based computations (by means of spikes) and sparsity, reducing the number of multiplications that are required. A recent study (St&#x000F6;ckl and Maass, <xref ref-type="bibr" rid="B36">2021</xref>) illustrated these potential SNN energy savings.</p>
<p>Recent years have shown great progress in learning algorithms for deep SNN. Especially training SNN, based on backpropagation-through-time with surrogate gradients (Neftci et al., <xref ref-type="bibr" rid="B25">2019</xref>), helped overcome the problem of the non-differentiability, which was introduced by the thresholding mechanism in a spiking neuron. These advances enabled SNN to move to deeper and more complicated model architectures using attention mechanisms (Yao et al., <xref ref-type="bibr" rid="B43">2023</xref>) or transformers (Zhou et al., <xref ref-type="bibr" rid="B49">2023</xref>; Zhu et al., <xref ref-type="bibr" rid="B50">2023</xref>). The main issue related to SNN however persists: frequently the SNN model does not perform as well as the equivalent ANN.</p>
<p>In biology, researchers have found that the brain is equipped with a plethora of powers to process spike trains adequately. One of those is the axonal delay. The transmission speed of an action potential is known to depend on the myelination of the axon and thus determine the extent to which a spike is delayed (Purves et al., <xref ref-type="bibr" rid="B31">2001</xref>). Furthermore, these delays are crucial in sensory processing (Orchard and Etienne-Cummings, <xref ref-type="bibr" rid="B26">2014</xref>) and known to adapt during the learning process (Lin and Faber, <xref ref-type="bibr" rid="B22">2002</xref>). A delay-enabled spiking network was also found to be able to compute a richer class of functions than a threshold circuit with adjustable weights (Maass and Schmitt, <xref ref-type="bibr" rid="B24">1999</xref>). Another differentiator between classical ANN and the brain is the widespread heterogeneity and neuronal adaptation processes taking place. Neurons of all forms and shapes are found, enabling a wide array spike pattern processing functions (Gerstner and Kistler, <xref ref-type="bibr" rid="B14">2002</xref>). Moreover, biological neurons exhibit slow dynamical processes that act at longer time scales, enabling processing of events that are temporally distanced in an implicit manner. These adaptive processes often limit the number of spikes produced.</p>
<p>Combining the ability to optimize the weights, delays and training neuronal parameters can lead to a more diverse and possibly improved internal representation. <xref ref-type="fig" rid="F1">Figure 1</xref> shows the responses of (A) a typical leaky integrate-and-fire (LIF) neuron, (B) a neuron with delayed input spike trains, and (C) a neuron with delayed input spike trains and neuronal adaptation for a spike train of four equidistant input spikes and all equal connection weights. It can clearly be observed that the output spike patterns are vastly different. The LIF neuron (A) will always respond the same while the others provide a wider range of possible results because of the trainable delay (B) and non-linear, trainable adaptation processes (C). Both extensions show to be complementary in providing additional memory. The synaptic delay allows the neuron to explicitly correlate incoming spikes at longer timescales, whereas the adaptation implicitly alters the behavior of a neuron based on its past regime.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Illustration of the variety in responses for different neuron models: two input spike trains are processed by three different neurons with equal weights. Input neuron 1 spikes at (<italic>t</italic><sub>2</sub>, <italic>t</italic><sub>4</sub>) and input neuron 2 spikes at (<italic>t</italic><sub>1</sub>, <italic>t</italic><sub>3</sub>). For all neurons, we show the evolution of the membrane potential, u[t] over time in response to these input spikes: <bold>(A)</bold> A typical leaky-integrate-and-fire (LIF) neuron, spikes at the timestep of the last incoming spike, <italic>t</italic><sub>4</sub>. <bold>(B)</bold> A LIF neuron with delayed (by <italic>d</italic>) input spikes from input neuron 2 produces spikes at timesteps <italic>t</italic><sub>2</sub> and <italic>t</italic><sub>4</sub>. <bold>(C)</bold> An adaptive neuron processes delayed input spikes from input neuron and only produces a spike at timestep <italic>t</italic><sub>3</sub> &#x0002B; <italic>d</italic>, the latency of the spikes coming from neuron 2.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-18-1360300-g0001.tif"/>
</fig>
<p>In this study, we present an SNN with adaptive neurons and synaptic delays that are co-optimized. Whereas, normally, as in ANN, just the synaptic weights are trained, we show that co-learning the delays and adaptation parameters individually enhance the performance of the SNN model, and that combining them even leads to state-of-the-art SNN results. The contribution of this study is summarized as follows:</p>
<list list-type="order">
<list-item><p>We analyze the biologically plausible neuronal adaptation parameters and which effects parameter boundaries have on the neurons working regime as well as on the SNN performance on three speech recognition datasets.</p></list-item>
<list-item><p>We introduce a novel learning rule for synaptic delays, which accounts for temporal context.</p></list-item>
<list-item><p>We present a novel SNN model in which both synaptic weights and delays are co-optimized in collaboration with the neuronal adaptation parameters.</p></list-item>
<list-item><p>We show that the inclusion of these more complex neurons through adaptation and the addition of trainable synaptic delays for every synapse specifically leads to state-of-the-art results for spiking neural networks. The proposed SNN even outperforms its non-spiking counterpart with equivalent model size on the speech recognition problems.</p></list-item>
</list></sec>
<sec id="s2">
<title>2 Related work</title>
<sec>
<title>2.1 Learning algorithms for spiking neural networks</title>
<p>Recently, there has been a rapid evolution in the development and progress of SNN learning paradigms. In general, either a trained ANN is converted into a rate-based SNN, aiming at minimal performance losses due to the ANN-SNN conversion (Deng and Gu, <xref ref-type="bibr" rid="B8">2021</xref>; Bu et al., <xref ref-type="bibr" rid="B3">2022</xref>) or the SNN is trained directly as a spiking neural network. A directly trained SNN is typically trained with backpropagation-through-time with surrogate gradients (Neftci et al., <xref ref-type="bibr" rid="B25">2019</xref>; Zenke and Vogels, <xref ref-type="bibr" rid="B47">2021</xref>). These surrogates are used to approach the derivative of non-differential Heaviside function, which is introduced by the spiking mechanism. A recent study went one step further in using differentiable spikes (Li et al., <xref ref-type="bibr" rid="B21">2021</xref>) for temporal credit assignment. Moreover, the spike-element-wise (SEW) ResNet (Fang et al., <xref ref-type="bibr" rid="B10">2021a</xref>) was proposed for training SNN without the vanishing/exploding problem introduced by surrogate gradients, paving the way for deeper SNN models.</p></sec>
<sec>
<title>2.2 Learning of neuronal parameters and neuronal heterogeneity</title>
<p>Another trend in SNN research is to learn the optimal distribution of the neuronal (leakage) parameters (Fang et al., <xref ref-type="bibr" rid="B11">2021b</xref>; Yin et al., <xref ref-type="bibr" rid="B45">2021</xref>; Rathi and Roy, <xref ref-type="bibr" rid="B32">2023</xref>). Moreover, the Gated LIF neuron (Yao et al., <xref ref-type="bibr" rid="B44">2022</xref>) was proposed to control the fusion of learnable membrane-related parameters. Additionally, in many studies, the heterogeneity of neurons in SNN proved to be beneficial for improved recognition performance (Perez-Nieves et al., <xref ref-type="bibr" rid="B30">2021</xref>; Deckers et al., <xref ref-type="bibr" rid="B7">2022</xref>; Chakraborty and Mukhopadhyay, <xref ref-type="bibr" rid="B4">2023</xref>).</p>
<p>Different methods for including neuronal adaptation processes have been proposed. The first class contains neurons with an adaptive threshold, which is increased after every spike and exponentially decays over time (Salaj et al., <xref ref-type="bibr" rid="B33">2021</xref>; Yin et al., <xref ref-type="bibr" rid="B45">2021</xref>). Others (Falez et al., <xref ref-type="bibr" rid="B9">2019</xref>) proposed a method for tuning the thresholds with specific target spike timestamps as an objective. In other studies (Gast et al., <xref ref-type="bibr" rid="B13">2020</xref>), the adaptive thresholds were combined with a synaptic depression model, which showed to replicate both bursting and steady-state behavior. Another adaptation method is based on adaptation currents, coupling a secondary variable to the sub-threshold membrane potential and its spike activity (Brunel et al., <xref ref-type="bibr" rid="B2">2003</xref>). This method was successfully formalized into the AdLIF spiking neuron model (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>) for SNN and was shown to outperform adaptive threshold-based models.</p></sec>
<sec>
<title>2.3 Delays in SNN</title>
<p>Many methods have been proposed for adapting propagation delays, inspired by spike timing dependent plasticity (Wang et al., <xref ref-type="bibr" rid="B40">2013</xref>) or based on the ReSuMe learning rule (Zhang et al., <xref ref-type="bibr" rid="B48">2020</xref>). A method for training per neuron axonal delays based on the SLAYER learning paradigm was proposed (Shrestha and Orchard, <xref ref-type="bibr" rid="B34">2018</xref>) and extended (Sun et al., <xref ref-type="bibr" rid="B38">2023b</xref>) with trainable delay caps. Recently, the effects of axonal synaptic delay learning were studied by pruning multiple delay synapses (Pati&#x000F1;o-Saucedo et al., <xref ref-type="bibr" rid="B29">2023</xref>), modeling a one-layer multinomial logistic regression with synaptic delays (Grimaldi and Perrinet, <xref ref-type="bibr" rid="B18">2023</xref>) and learning delays represented trough 1D convolutions with learnable spacings (Hammouamri et al., <xref ref-type="bibr" rid="B19">2023</xref>). Similarly, in order to train synaptic delays, spike trains were transformed into continuous analog, differentiable signals (Wang et al., <xref ref-type="bibr" rid="B41">2019</xref>). Surprisingly, only learning the delays (Grappolini and Subramoney, <xref ref-type="bibr" rid="B17">2023</xref>) showed to achieve comparable performance, only learning the weights.</p></sec></sec>
<sec sec-type="materials and methods" id="s3">
<title>3 Materials and methods</title>
<p>A spiking neural network (SNN) is a biologically inspired type of neural network, in which spikes, i.e., binary events are used to communicate between layers of spiking neurons. In this section, we elaborate on the fundamental properties of spiking neurons and the methods used in this study to train the synaptic weights, synaptic delays, and neuronal parameters in multi-layer networks of spiking neurons.</p>
<sec>
<title>3.1 Spiking neurons</title>
<p>Spiking neurons differ from classical neurons in ANN because of their inherent time-dependent processing of data streams. Incoming spikes are multiplied by the synaptic weights and accumulated over time for every neuron. When this neuronal state, i.e., the membrane potential crosses the spiking threshold, a neuron emits a spike to a subsequent layer. The membrane potential is maintained over time and thus creates an internal memory for every individual neuron.</p>
<p>In this study, we use the adaptive leaky integrate-and-fire neuron (AdLIF) model (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>) with updated parameter boundaries. To highlight this modification, our neuron model is called the constrained AdLIF, (cAdLIF). In this model, two internal states are kept: the membrane potential and the adaptation current, which provides the neuronal adaptation. Formally, <italic>u</italic>[<italic>t</italic>], <italic>w</italic>[<italic>t</italic>], and <italic>s</italic>[<italic>t</italic>], respectively, represent the membrane potential, the adaptation current, and the presence of a spike, at time step t. In the AdLIF neuron model, there are four trainable neuronal parameters: &#x003B1; and &#x003B2; denote the leak of u[t] and w[t], respectively, while <italic>a</italic> and <italic>b</italic> describe the characteristics of the adaptation current. The adaptation current is coupled with the sub-threshold membrane potential by <italic>a</italic>, while <italic>b</italic> represents the spike-triggered adaptation. In the cAdLIF model, these <italic>a</italic> and <italic>b</italic> are constrained to be positive only. This change results in major differences when processing spikes, as discussed in Section 4.3. A neuron generates a spike at timestep t when the membrane potential crosses the firing threshold, &#x003B8;, at t. Mathematically, the threshold &#x003B8; is represented as a Heaviside function. The full discrete-time model, with time step size equal to 1 ms, is described in <xref ref-type="disp-formula" rid="E1">Equation (1)</xref>. The trainable neuronal parameters are highlighted in bold.</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>u</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi mathvariant="bold-italic">&#x003B1;</mml:mi><mml:mi>u</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi mathvariant="bold-italic">&#x003B1;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>I</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>w</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>&#x003B8;</mml:mi><mml:mi>s</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>w</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi mathvariant="bold-italic">&#x003B2;</mml:mi><mml:mi>w</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi mathvariant="bold-italic">&#x003B2;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext class="textit" mathvariant="bold-italic">a</mml:mtext><mml:mi>u</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mtext class="textit" mathvariant="bold-italic">b</mml:mtext><mml:mi>s</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>s</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>u</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x02265;</mml:mo><mml:mi>&#x003B8;</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The adaptation, implemented by means of this adaptation current <italic>w</italic>[<italic>t</italic>] is affected by both the neurons&#x00027; instantaneous membrane potential and a spike-triggered fraction. This contrasts with the adaptive neuronal threshold adaptation, in which only the spiking activity is taken into account (Yin et al., <xref ref-type="bibr" rid="B45">2021</xref>). This model was chosen because of its proven superior performance in comparison with a leaky integrate-and-fire (LIF) model with an adaptive neuronal threshold (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>). The computational graph of the neuron model, rolled out over time, is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. The yellow box, <italic>w</italic>[<italic>t</italic>], represents the addition of the adaptation variable to the classic LIF neuron model, and the blue arrows, connecting the internal variables, represent the corresponding trainable neuron parameters.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p><bold>(Left)</bold> Computational graph of the AdLIF neuron model, unrolled over time. The yellow blocks denote the additional adaptation current parameter, which depends on the membrane potential u[t] and the spike activity, s[t] at the previous timestep. <bold>(Right)</bold> Two-layer fully connected architecture with readout of the output neuron membrane potential, unrolled over the time. The blue connections denote the potential delays in the network. For clarity, the delays starting from time step 1 onward were omitted.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-18-1360300-g0002.tif"/>
</fig></sec>
<sec>
<title>3.2 Training multi-layer SNN</title>
<sec>
<title>3.2.1 Model architecture</title>
<p>In this study, the SNN consists of a simple feed-forward network with two hidden layers. <xref ref-type="fig" rid="F2">Figure 2</xref> shows the network architecture, which is rolled out over time. In general, neuron <italic>i</italic> in hidden layer <italic>l</italic> receives <inline-formula><mml:math id="M2"><mml:msubsup><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, the pre-synaptic current, which consists of two elements, as shown in <xref ref-type="disp-formula" rid="E2">Equation (2)</xref>. The feed-forward synapses from the previous layer <italic>l</italic>&#x02212;1 have associated weights <inline-formula><mml:math id="M3"><mml:msubsup><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> and carry spikes from the same time step <italic>t</italic> to neuron j and the neuronal bias, <inline-formula><mml:math id="M4"><mml:msubsup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>. These input spike trains are summed over all pre-synaptic neurons <italic>j</italic> &#x0003D; 1, ..., <italic>N</italic><sub><italic>l</italic>&#x02212;1</sub>. Spike trains are represented by <italic>s</italic>[<italic>t</italic>] &#x02208; [0, 1]. Typically, in this type of architecture, all feed-forward are connected in an all-to-all fashion.</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The readout mechanism, which is used to derive the outputs of the SNN, consists of a single layer. This layer consists of neurons with infinite threshold. These neurons have no memory, and hence, the membrane potential is equal to the the weighted inputs. The output of the SNN model is the sum of the membrane potential of the output neurons over time, which is passed though a softmax layer for every timestep. The outputs are shown in <xref ref-type="disp-formula" rid="E3">Equation (3)</xref>.</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mstyle><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></sec>
<sec>
<title>3.2.2 Training procedure</title>
<p>Typically, spiking neural networks (SNN) are trained via backpropagation-through-time (BPTT) with surrogate gradients (Neftci et al., <xref ref-type="bibr" rid="B25">2019</xref>). In these methods, the summed membrane potential of the output neurons, see <xref ref-type="disp-formula" rid="E3">Equation (3)</xref>, constitute the cross-entropy loss of the network, which is unrolled over time. The loss with respect to class c for a batch size <italic>N</italic> is represented as follows:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo>-</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mstyle><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Based on the chain rule in error-backpropagation, the weight update for neuron <italic>i</italic> in the penultimate layer <italic>l</italic> for a sequence of T timesteps is shown in <xref ref-type="disp-formula" rid="E5">Equation (5)</xref>.</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In these methods, surrogate gradients are used to approximate the non-differentiability, which was introduced by the thresholding mechanism (Heaviside function), <inline-formula><mml:math id="M9"><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:mi>s</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:mi>u</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:math></inline-formula>, with a differentiable function in the backward pass of error-backpropagation. For simplicity and comparability with the previous study (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>), the boxcar surrogate gradient function, shown in <xref ref-type="disp-formula" rid="E6">Equation (6)</xref>, was used in this study.</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M10"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:mi>s</mml:mi><mml:mo stretchy='false'>[</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>]</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:mi>u</mml:mi><mml:mo stretchy='false'>[</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>]</mml:mo></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo> <mml:mrow><mml:mtable columnalign='left'><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mrow><mml:mn>0.5</mml:mn></mml:mrow></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>if&#x000A0;</mml:mtext><mml:mo>&#x0007C;</mml:mo><mml:mi>u</mml:mi><mml:mo stretchy='false'>[</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>]</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B8;</mml:mi><mml:mtext>&#x0007C;</mml:mtext><mml:mo>&#x02264;</mml:mo><mml:mtext>0</mml:mtext><mml:mo>.</mml:mo><mml:mtext>5</mml:mtext></mml:mrow></mml:mtd></mml:mtr><mml:mtr columnalign='left'><mml:mtd columnalign='left'><mml:mn>0</mml:mn></mml:mtd><mml:mtd columnalign='left'><mml:mrow><mml:mtext>otherwise</mml:mtext></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow> </mml:mrow></mml:mrow></mml:math></disp-formula>
</sec></sec>
<sec>
<title>3.3 Introducing synaptic delays</title>
<p>In contrast to ANNs and most other SNNs in which connections are defined by a weight parameter, in this study, connections between neurons are characterized by two synaptic parameters: a weight and a delay. The addition of synaptic delays leads to an adjusted neuronal processing model. Now, the spike train for neuron <italic>i</italic> from layer <italic>l</italic> is characterized by <inline-formula><mml:math id="M11"><mml:msubsup><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, as shown in <xref ref-type="disp-formula" rid="E7">Equation (7)</xref>. The synaptic delay can be found in <inline-formula><mml:math id="M12"><mml:msubsup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>, where <italic>d</italic><sub><italic>ij</italic></sub> denotes the delayed spikes between neuron j and i.</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Training of the synaptic delays is an adaptation of the SLAYER learning method (Shrestha and Orchard, <xref ref-type="bibr" rid="B34">2018</xref>) for training axonal delays, which is applied to individual synapses based on the local temporal context. The delay kernel, &#x003F5;<sub><italic>d</italic></sub>, is convolved with spike train s[t], to get the delayed spike kernel, <inline-formula><mml:math id="M14"><mml:msup><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003F5;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>*</mml:mo><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>. Similarly to the case in which delays were equal to 0 (i.e., without delays), the derivative of the loss with respect to the synaptic weight and delays is now computed, as shown in <xref ref-type="disp-formula" rid="E8">Equations (8</xref>, <xref ref-type="disp-formula" rid="E9">9)</xref>.</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E9"><label>(9)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In <xref ref-type="disp-formula" rid="E9">Equation (9)</xref>, <inline-formula><mml:math id="M17"><mml:mfrac><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></inline-formula> is equal to <inline-formula><mml:math id="M18"><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula> for every unique synapse. Contrary to the SLAYER delay learning algorithm as presented in the study by Shrestha and Orchard (<xref ref-type="bibr" rid="B34">2018</xref>), where the finite difference instantaneous derivative with respect to time is taken or Sun et al. (<xref ref-type="bibr" rid="B37">2023a</xref>) where the axonal, i.e., identical for every postsynaptic neuron, delay is implemented as a variable axonal delay module, we take the temporal context into account for every individual synapse. More precisely, <inline-formula><mml:math id="M19"><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula> is replaced with <italic>p</italic>[<italic>t</italic> &#x02212; <italic>c</italic>:<italic>t</italic> &#x02212; 1] &#x02212; <italic>p</italic>[<italic>t</italic> &#x0002B; 1:<italic>t</italic> &#x0002B; <italic>c</italic>] for all t timesteps per synapse. In this way, the direction of the delay updates is determined by the corresponding temporal context <italic>c</italic> both in the past and future of a specific timestep t and not just the direct derivative at <italic>t</italic>. We experimented with different ranges of temporal contexts in the learning rule for synaptic delays. To this end, we compared different values of <italic>c</italic>. We empirically found a width of 3 to be optimal. A detailed analysis is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Illustration of different neuron model adaptation parametrizations and their responses to a fixed train of 12 incoming spikes. We analyze the number of output spikes and the corresponding evolution of the membrane potential u[t] over time for a range of <italic>a</italic> and <italic>b</italic> neuronal adaptation parameters. A model without adaptation, the LIF model, is found where both <italic>a</italic> and <italic>b</italic> are equal to zero. The AdLIF model, which allows both positive and negative <italic>a</italic>, can possibly result in an unstable spiking regime. In this case, the neuron generates more spikes than the number of incoming spikes, 12 in this example. We therefore constrained the updated neuron model, cAdLIF to remain within positive <italic>a</italic> and <italic>b</italic> boundaries. For reference, we also show the parametrizations for 6 and 1 output spikes in purple and green, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-18-1360300-g0003.tif"/>
</fig>
</sec></sec>
<sec id="s4">
<title>4 Experiments</title>
<p>In this section, we will elaborate on the experiments conducted for this study and the corresponding results. First, we describe the datasets and the precise setup for training the SNNs. Following this, we present an analysis of the AdLIF neuron model and, finally, show the full results.</p>
<sec>
<title>4.1 Datasets</title>
<p>We used three common SNN benchmark speech recognition datasets: the Spiking Heidelberg Digits (SHD), the Spiking Speech Commands (SSC) (Cramer et al., <xref ref-type="bibr" rid="B5">2020</xref>), and the Google Speech Commands v0.02 (GSC) dataset (Warden, <xref ref-type="bibr" rid="B42">2018</xref>). The first two datasets are neuromorphic datasets, in which the original sounds were converted to spikes, spread out over 700 input channels. The GSC dataset consists of speech samples. The SHD dataset consists of German and English spoken digits (0 through 9). The SSC dataset is based on the sounds from the GSC dataset. Both the SSC and GSC datasets have 35 classes from a large group of speakers in a non-controlled environment. These larger datasets provide a more challenging speech recognition task.</p>
<p>Similarly to other studies, for the spiking datasets, which were zero-padded and aligned to 1 s, we binned the input data from 700 input channels to 140 channels and 100 timesteps with 10 ms bins to ensure uniformity across all samples. Regarding the GSC dataset, speech data were aligned to 1 s by padding with zeros and thereafter binned in 10 millisecond bins to generate samples of 100 timesteps. Further processing was performed with a Mel filterbank with 40 Mel filters. Since there is no predefined test set for the SHD dataset, we decided to take the average accuracy on the validation set across 10 experiments with different random seeds. Similarly, three random trials were conducted for the SSC and GSC datasets.</p></sec>
<sec>
<title>4.2 Training setup</title>
<p>All neuronal trainable parameters were uniformly initialized between specific boundaries and subsequently co-learned with the other model trainable parameters to reflect the neuronal heterogeneity (Perez-Nieves et al., <xref ref-type="bibr" rid="B30">2021</xref>). These parameters were initialized following a uniform distribution: &#x003B1; &#x02208; [0.36, 0.96], &#x003B2; &#x02208; [0.96, 0.99], <italic>a</italic> &#x02208; [0, 1], and <italic>b</italic> &#x02208; [0, 2]. Different from the AdLIF model (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>), we extended the available range of the membrane potential decay parameter &#x003B1; and limited the dependency of the adaptation current with respect to the membrane potential, <italic>a</italic>, to remain positive, which showed to stabilize the neuron model for sparse input data. During the training process, all neuron parameters are clipped to remain within these boundaries. The spike threshold was fixed at 1 (dimensionless). The weights of all connections were initialized, following the default Xavier uniform distribution, and the neuronal biases were set to zero. For all hidden neurons, the initial membrane potential and adaptation current were zero-initialized.</p>
<p>We used the Adam optimizer (Kingma and Ba, <xref ref-type="bibr" rid="B20">2014</xref>) in all experiments with an initial learning rate of 0.01 for the SHD dataset and 0.001 for the others, which is similar to the previous study (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>). The learning rate for the delays was always equal to 10&#x0002A;<italic>lr</italic><sub><italic>weigths</italic></sub>. We used a simple scheduler for both the weights and delays, which decreased the learning rate by a factor of 0.7 with a patience of 5 epochs. Dropout (Srivastava et al., <xref ref-type="bibr" rid="B35">2014</xref>) was applied in the hidden layers with rates of 0.5 and 0.25 for SHD and the other datasets, respectively. The available delay values are limited to remain within [0, 25] time steps for these datasets. The network is initialized without delays, i.e., all delays are equal to zero, and the input data were right padded accordingly for the network to allow delayed spikes. This exact setup was used for all experiments on all datasets presented in this study and implemented in PyTorch (Paszke et al., <xref ref-type="bibr" rid="B28">2019</xref>). Our code is based on the SpArch implementation (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>). <xref ref-type="table" rid="T1">Table 1</xref> shows an overview of the exact parametrization of the experiments, executed for this study. The setups for the SSC and GSC datasets were identical.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Overview of the hyperparameters used in training for all datasets considered in this study.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919497;color:#ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Hidden layer size</bold></th>
<th valign="top" align="center"><bold>Epochs</bold></th>
<th valign="top" align="center"><bold>Batch size</bold></th>
<th valign="top" align="center"><bold>Dropout rate</bold></th>
<th valign="top" align="center"><bold>Lr weights</bold></th>
<th valign="top" align="center"><bold>Lr delays</bold></th>
<th valign="top" align="left"><bold>Initialization</bold></th>
<th valign="top" align="center"><bold>Delay caps</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SHD</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">100</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">0.5</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="center">0.1</td>
<td valign="top" align="left">Xavier uniform</td>
<td valign="top" align="center">[0, 25]</td>
</tr>
<tr>
<td valign="top" align="left">SSC &#x00026; GSC</td>
<td valign="top" align="center">512</td>
<td valign="top" align="center">100</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">0.25</td>
<td valign="top" align="center">0.001</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="left">Xavier uniform</td>
<td valign="top" align="center">[0, 25]</td>
</tr></tbody>
</table>
</table-wrap></sec>
<sec>
<title>4.3 Analysis of the trainable adaptation parameters</title>
<p>In this section, we analyze the effects of the different parameter ranges for the main adaptation parameters <italic>a</italic> and <italic>b</italic> from the AdLIF neuron model, as presented in <xref ref-type="disp-formula" rid="E1">Equation (1)</xref>. Our constrained Adaptive LIF (cAdLIF) model differs from the AdLIF model in two ways: we extended the available decay window for the membrane potential, the &#x003B1; parameter, for the model to be able to forget more quickly if needed. More importantly, we limited the available range of the <italic>a</italic> parameter, which allows the current membrane potential u[t] to influence the adaptation current w[t]. Similarly to adaptation by means of an adaptive threshold, this <italic>a</italic> parameter mainly limits spikes if the membrane potential was high in the past and thus provides a homeostatic mechanism, given that this <italic>a</italic> remains positive.</p>
<p>In <xref ref-type="fig" rid="F3">Figure 3</xref>, we illustrate that for a possible negative <italic>a</italic>, the behavior of the adaptation current can lead to a non-controlled, chaotic spiking regime. In this figure, we count the number of spikes that are produced by a neuron, given a single fixed input spike train of 12 spikes, for different adaptation parametrizations of <italic>a</italic> and <italic>b</italic>. Whenever the produced number of spikes is higher than the number of input spikes, the model enters a chaotic regime in which a positive feedback loop could be activated and the neuron remains spiking, even without the presence of input spikes. <xref ref-type="fig" rid="F3">Figure 3</xref> shows that in the case for a negative <italic>a</italic>, where, apart from a region with a very high parameter <italic>b</italic>, the spike-triggered fraction of the adaptation current, the number of output spikes generated is very high. We therefore, unlike other works, chose to constrain in the top right quadrant with both positive <italic>a</italic> and <italic>b</italic> and hence the constrained AdLIF (cAdLIF) name for our neuron model. In this figure, the base LIF model can be found at the point, where <italic>a</italic> and <italic>b</italic> are equal to 0, and no adaptation current w[t] is produced.</p></sec>
<sec>
<title>4.4 Results</title>
<sec>
<title>4.4.1 Speech recognition datasets</title>
<p>In the first experiment, we investigated the effects of adding the trainable parameters to a basic LIF neuron model with similar initialization and the effects of updated parametrization boundaries for the adaptation parameters. Secondly, we added the trainable synaptic delays (models with d-) and evaluated the interplay of co-learning the adaptation and delays. The results in terms of classification accuracy, highlighting the individual contributions, are shown in <xref ref-type="table" rid="T2">Table 2</xref>, as tested on the three speech recognition datasets.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Effect of the proposed enhancements in terms of accuracy, number of trainable parameters, and average spikes per neuron per sample, to a basic 2-hidden layer feedforward SNN model with a hidden layer size of 128.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919497;color:#ffffff">
<th/>
<th/>
<th valign="top" align="center"><bold>LIF</bold></th>
<th valign="top" align="center"><bold>AdLIF</bold></th>
<th valign="top" align="center"><bold>cAdLIF</bold></th>
<th valign="top" align="center"><bold>d-LIF</bold></th>
<th valign="top" align="center"><bold>d-AdLIF</bold></th>
<th valign="top" align="center"><bold>d-cAdLIF</bold></th>
<th valign="top" align="center"><bold>fd-cAdLIF</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="3">Accuracy (%)</td>
<td valign="top" align="center">SHD</td>
<td valign="top" align="center">84.49</td>
<td valign="top" align="center">91.67</td>
<td valign="top" align="center">94.19</td>
<td valign="top" align="center">92.57</td>
<td valign="top" align="center">93.40</td>
<td valign="top" align="center">94.85</td>
<td valign="top" align="center">93.40</td>
</tr>
 <tr>
<td valign="top" align="center">SSC</td>
<td valign="top" align="center">71.76</td>
<td valign="top" align="center">76.75</td>
<td valign="top" align="center">77.5</td>
<td valign="top" align="center">75.94</td>
<td valign="top" align="center">78.9</td>
<td valign="top" align="center">80.23</td>
<td valign="top" align="center">77.72</td>
</tr>
 <tr>
<td valign="top" align="center">GSC</td>
<td valign="top" align="center">86.21</td>
<td valign="top" align="center">93.97</td>
<td valign="top" align="center">94.67</td>
<td valign="top" align="center">89.81</td>
<td valign="top" align="center">95.3</td>
<td valign="top" align="center">95.69</td>
<td valign="top" align="center">95.05</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">&#x00023; Parameters</td>
<td valign="top" align="center">SHD</td>
<td valign="top" align="center">37.9 k</td>
<td valign="top" align="center">38.7 k</td>
<td valign="top" align="center">38.7 k</td>
<td valign="top" align="center">74.8 k</td>
<td valign="top" align="center">75.8 k</td>
<td valign="top" align="center">75.8 k</td>
<td valign="top" align="center">38.7 k</td>
</tr>
 <tr>
<td valign="top" align="center">SSC</td>
<td valign="top" align="center">0.34 M</td>
<td valign="top" align="center">0.35 M</td>
<td valign="top" align="center">0.35 M</td>
<td valign="top" align="center">0.69 M</td>
<td valign="top" align="center">0.7 M</td>
<td valign="top" align="center">0.7 M</td>
<td valign="top" align="center">0.35 M</td>
</tr>
 <tr>
<td valign="top" align="center">GSC</td>
<td valign="top" align="center">0.30 M</td>
<td valign="top" align="center">0.30 M</td>
<td valign="top" align="center">0.30 M</td>
<td valign="top" align="center">0.60 M</td>
<td valign="top" align="center">0.61 M</td>
<td valign="top" align="center">0.61 M</td>
<td valign="top" align="center">0.30 M</td>
</tr> <tr>
<td valign="top" align="left" rowspan="3">&#x00023;spikes/neuron</td>
<td valign="top" align="center">SHD</td>
<td valign="top" align="center">5.4</td>
<td valign="top" align="center">5.7</td>
<td valign="top" align="center">5.6</td>
<td valign="top" align="center">4.7</td>
<td valign="top" align="center">9,8</td>
<td valign="top" align="center">5.8</td>
<td valign="top" align="center">6.1</td>
</tr>
 <tr>
<td valign="top" align="center">SSC</td>
<td valign="top" align="center">6.6</td>
<td valign="top" align="center">14.6</td>
<td valign="top" align="center">3.9</td>
<td valign="top" align="center">7.7</td>
<td valign="top" align="center">9.2</td>
<td valign="top" align="center">5.2</td>
<td valign="top" align="center">6.2</td>
</tr>
<tr>
<td valign="top" align="center">GSC</td>
<td valign="top" align="center">12.5</td>
<td valign="top" align="center">13.1</td>
<td valign="top" align="center">5.0</td>
<td valign="top" align="center">10.1</td>
<td valign="top" align="center">7.6</td>
<td valign="top" align="center">6.7</td>
<td valign="top" align="center">8.6</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Given the limited size of the SHD validation set, the average accuracy over 10 runs is shown. The constrained AdLIF with trainable delays (d-cAdLIF) shows the best performance.</p>
</table-wrap-foot>
</table-wrap>
<p>First, comparing the LIF model with the AdLIF and the cAdLIF neuron models on the SHD dataset, we observed that the average validation accuracy over 10 experiments is significantly increased by up to 7.2%. The cAdLIF model performs &#x0007E;2.5% better than the AdLIF model, on average, which is in accordance with the empirical evaluation of adaptation parameters <italic>a</italic> and <italic>b</italic> in the analysis from Section 4.3. For the larger speech datasets, similar improvements are shown. Adding adaptation showed an increase of 5.0 and 8.1%, and the cAdLIF model further increased the accuracy with 0.75 and 0.7% for the SSC and GSC datasets, respectively. These results show the importance of trainable adaptation in SNN and adequate trainable neuron parameter boundaries. Similarly, the d-cAdLIF significantly outperforms the d-AdLIF and the d-LIF.</p>
<p>Secondly, the addition of trainable synaptic delays increases the average accuracy for the LIF, d-AdLIF, and cAdLIF neuron models by 8.1%, 1.7%, and 0.66% for the SHD dataset. The models with trainable synaptic delays are titled d-<italic>SNN</italic>. Comparably, enhanced results are shown for the SSC and GSC datasets. This shows that the extra provided capacity to utilize memory is beneficial for all speech recognition tasks and adds to the computational complexity already provided by the trained adaptation parameters. Additionally, co-learning both shows to further enhance our results. To validate the effectiveness of the proposed learning rule, we also trained an SNN with random heterogeneous, fixed (non-learnable) delays, called fd-cAdLIF. In this model, the delays were uniformly distributed within the pre-determined intervals ([0, 25]). The d-cAdLIF outperforms the fd-cAdLIF on all datasets.</p>
<p><xref ref-type="table" rid="T2">Table 2</xref> also shows the number of trainable parameters for all datasets and model configurations. The number of additional trainable parameters is limited when comparing the LIF with AdLIF and cAdLIF models as they only increase with the number of neurons in the SNN model not the synapses. Combined with the results in classification, these results clearly show the added value of trainable adaptation parameters for SNN models. Logically, the number of trainable parameters is doubled for the delay-enabled SNNs, as in that case, every synapse is characterized by both a delay and a weight value.</p>
<p>One of the properties that affects the efficiency of the SNN model, when deployed on neuromorphic hardware Yin et al. (<xref ref-type="bibr" rid="B45">2021</xref>), is the number of spikes that are required when processing a single sample in inference. We therefore analyze the number of spikes in the hidden layers to assess how the proposed improvements influence the total number of spikes in the SNN. To account for inter-experimental variance, we averaged the spike rates over 10 trials on the SHD dataset. The results are shown in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<p>We observed that the number of spikes is not significantly increased by the addition of the proposed adaptation method. For the SSC and GSC datasets, there is a decrease in the number of spikes, especially comparing the AdLIF and cAdLIF models. This shows that BPTT effectively ends up in regions where <italic>a</italic> or <italic>b</italic> &#x0003C; 0 and uncontrolled spiking behavior occur with the unconstrained AdLIF model. However, the synaptic delays, in general, result in an increase in the number of spikes in the SNN model.</p></sec>
<sec>
<title>4.4.2 Experimental analysis on SHD</title>
<p>In <xref ref-type="fig" rid="F4">Figure 4</xref>, we show a more detailed analysis of the SNN models which are trained on SHD. In <xref ref-type="fig" rid="F4">Figure 4A</xref>, the distribution of the accuracy, as tested across 10 independent trials, is shown. The d-cAdLIF model clearly outperforms the cAdLIF model without synaptic delays and one with random fixed delays, showing the benefits of co-learning the adaptation and synaptic delays.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Analysis of the experiments on the SHD dataset. <bold>(A)</bold> Overview of the classification results for 10 independent trials for all evaluated neuron models. <bold>(B)</bold> Overview of the learned neuronal parameters in hidden layer 1 (top) and hidden layer 2 (bottom) for a d-cAdLIF model. <bold>(C)</bold> Experiments on the temporal delay parameter c, for 10 independent trials. <bold>(D)</bold> distribution of the learned synaptic delays per layer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-18-1360300-g0004.tif"/>
</fig>
<p><xref ref-type="fig" rid="F4">Figure 4B</xref> shows the distribution of the trained neuronal parameters. Interestingly, these are relatively similar for the two hidden layers. We note that specifically <italic>a</italic> and <italic>b</italic>, although uniformly initialized between their respective boundaries, show large proportion of near-zero elements.</p>
<p>In <xref ref-type="fig" rid="F4">Figure 4C</xref>, we outlined the experiments on the temporal context <italic>c</italic>, which was used in training in the backward pass to determine the delay value updates. The results shown are for 10 independent d-cAdLIF neuron trials. We found that taking into account three timesteps in the past/future yielded the best average results. Finally, <xref ref-type="fig" rid="F4">Figure 4D</xref> shows the distribution of the learned delay values for all three layers. We again note that many values are close to zero.</p></sec>
<sec>
<title>4.4.3 Comparison to the state-of-the-art</title>
<p>An overview of the results of our experiments on the full cAdLIF model with synaptic delays across all speech recognition datasets is presented in <xref ref-type="table" rid="T3">Table 3</xref>. We compared our SNN model with state-of-the-art SNN solutions from the literature with a 2-hidden layer architecture and a corresponding state-of-the-art ANN model for each dataset, which is shown below the dotted line.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Test accuracy on the SHD, SSC, and GSC datasets and comparison with state-of-the-art in SNN for 2 hidden layer feedforward models, ordered for the number of trainable parameters.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919497;color:#ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Model</bold></th>
<th valign="top" align="center"><bold>Hidden size</bold></th>
<th valign="top" align="center"><bold>&#x00023; Parameters</bold></th>
<th valign="top" align="center"><bold>Accuracy (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="6">SHD</td>
<td valign="top" align="left">Adaptive RSNN (Yin et al., <xref ref-type="bibr" rid="B45">2021</xref>)</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">/</td>
<td valign="top" align="center">90.4</td>
</tr>
 <tr>
<td valign="top" align="center"><bold>d-cAdLIF (ours)</bold></td>
<td valign="top" align="center">128</td>
<td valign="top" align="center"><bold>0.076 M</bold></td>
<td valign="top" align="center">94.85 &#x000B1; 0.64</td>
</tr>
 <tr>
<td valign="top" align="left">Axonal delays (Sun et al., <xref ref-type="bibr" rid="B39">2022</xref>)</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">0.1 M</td>
<td valign="top" align="center">92.36</td>
</tr>
 <tr>
<td valign="top" align="left">Synaptic delays (Hammouamri et al., <xref ref-type="bibr" rid="B19">2023</xref>)</td>
<td valign="top" align="center">256</td>
<td valign="top" align="center">0.2 M</td>
<td valign="top" align="center">95.07 &#x000B1; 0.24</td>
</tr>
 <tr>
<td valign="top" align="left">DL256-SNN-DLoss (Sun et al., <xref ref-type="bibr" rid="B37">2023a</xref>)</td>
<td valign="top" align="center">256</td>
<td valign="top" align="center">0.21 M</td>
<td valign="top" align="center">93.55</td>
</tr>
 <tr>
<td valign="top" align="left">RadLIF (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>)</td>
<td valign="top" align="center">1,024</td>
<td valign="top" align="center">3.9 M</td>
<td valign="top" align="center">94.62</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">CNN (Cramer et al., <xref ref-type="bibr" rid="B5">2020</xref>)</td>
<td valign="top" align="center">/</td>
<td valign="top" align="center">/</td>
<td valign="top" align="center">92.4</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Adaptive RSNN (Yin et al., <xref ref-type="bibr" rid="B45">2021</xref>)</td>
<td valign="top" align="center">400</td>
<td valign="top" align="center">/</td>
<td valign="top" align="center">74.2</td>
</tr> <tr>
<td valign="top" align="left" rowspan="4">SSC</td>
<td valign="top" align="center"><bold>d-cAdLIF (ours)</bold></td>
<td valign="top" align="center">512</td>
<td valign="top" align="center"><bold>0.7 M</bold></td>
<td valign="top" align="center"><bold>80.23</bold> <bold>&#x000B1;</bold> <bold>0.07</bold></td>
</tr>
 <tr>
<td valign="top" align="left">Synaptic delays (Hammouamri et al., <xref ref-type="bibr" rid="B19">2023</xref>)</td>
<td valign="top" align="center">512</td>
<td valign="top" align="center">0.7 M</td>
<td valign="top" align="center">79.77 &#x000B1; 0.09</td>
</tr>
 <tr>
<td valign="top" align="left">Synaptic delays<sup>&#x0002A;&#x0002A;</sup> (Hammouamri et al., <xref ref-type="bibr" rid="B19">2023</xref>)</td>
<td valign="top" align="center">512</td>
<td valign="top" align="center">1.2 M</td>
<td valign="top" align="center">80.29 &#x000B1; 0.06</td>
</tr>
 <tr>
<td valign="top" align="left">RadLIF (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>)</td>
<td valign="top" align="center">1,024</td>
<td valign="top" align="center">3.9 M</td>
<td valign="top" align="center">77.4</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">GRU (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>)</td>
<td valign="top" align="center">512</td>
<td valign="top" align="center">/</td>
<td valign="top" align="center">79.05</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">RSNN, LIF (Zenke and Vogels, <xref ref-type="bibr" rid="B47">2021</xref>)</td>
<td valign="top" align="center">256</td>
<td valign="top" align="center">/</td>
<td valign="top" align="center">85.3</td>
</tr> <tr>
<td valign="top" align="left" rowspan="5">GSC</td>
<td valign="top" align="left">RSNN with SFA (Salaj et al., <xref ref-type="bibr" rid="B33">2021</xref>)</td>
<td valign="top" align="center">2,048<sup>&#x0002A;</sup></td>
<td valign="top" align="center">/</td>
<td valign="top" align="center">88.5</td>
</tr>
 <tr>
<td valign="top" align="center"><bold>d-cAdLIF (ours)</bold></td>
<td valign="top" align="center">512</td>
<td valign="top" align="center"><bold>0.61 M</bold></td>
<td valign="top" align="center"><bold>95.69</bold> <bold>&#x000B1;</bold> <bold>0.03</bold></td>
</tr>
 <tr>
<td valign="top" align="left">Synaptic delays (Hammouamri et al., <xref ref-type="bibr" rid="B19">2023</xref>)</td>
<td valign="top" align="center">512</td>
<td valign="top" align="center">0.7 M</td>
<td valign="top" align="center">94.91 &#x000B1; 0.09</td>
</tr>
 <tr>
<td valign="top" align="left">RadLIF (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>)</td>
<td valign="top" align="center">512</td>
<td valign="top" align="center">0.83 M</td>
<td valign="top" align="center">94.51</td>
</tr>
 <tr>
<td valign="top" align="left">Synaptic delays** (Hammouamri et al., <xref ref-type="bibr" rid="B19">2023</xref>)</td>
<td valign="top" align="center">512</td>
<td valign="top" align="center">1.2 M</td>
<td valign="top" align="center">95.29 &#x000B1; 0.11</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">GRU (Bittar and Garner, <xref ref-type="bibr" rid="B1">2022</xref>)</td>
<td valign="top" align="center">512</td>
<td valign="top" align="center">/</td>
<td valign="top" align="center">94.32</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Transformer (Gong et al., <xref ref-type="bibr" rid="B16">2021</xref>)</td>
<td valign="top" align="center">/</td>
<td valign="top" align="center">/</td>
<td valign="top" align="center">98.11</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>We report the average &#x000B1; std for all results. Benchmark ANN results are shown below the dotted line. Our model is highlighted in bold.</p>
<p><sup>&#x0002A;</sup>Recurrent SNN with a single hidden layer.</p>
<p><sup>&#x0002A;&#x0002A;</sup>Three hidden layers.</p>
</table-wrap-foot>
</table-wrap>
<p>For the SHD dataset, the cAdLIF model with trained synaptic delays matches the state-of-the-art results of a recently proposed alternative method for training synaptic delays at just a fraction (less than half) of its number of trainable parameters and outperforms all other SNN and ANN methods on this dataset. Here, the cAdLIF model with trainable synaptic delays shows better performance than the current state-of-the-art models in SNN with a similar number of trainable parameters or less. Furthermore, one additional SNN model with three hidden layers was proposed by Hammouamri et al. (<xref ref-type="bibr" rid="B19">2023</xref>). This model achieves similar performance as ours on the SSC dataset 80.29 &#x000B1; 0.06%, with an increase of &#x0007E;45% additional trainable parameters (an additional hidden layer). This model only achieved 95.29 &#x000B1; 0.11% on the GSC dataset, which is less than our SNN, which has just two hidden layers. Given the budget of trainable parameters, the proposed feed-forward SNN model even outperforms a non-SNN recurrent model (GRU) with the same preprocessing and moves closer to the performance of a large ANN model with transformer architecture.</p></sec></sec></sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>In this study, we presented a novel SNN model, the cAdLIF with a novel temporal context-aware learning rule for synaptic delays. To the best of our knowledge, this is the first SNN model in which the synaptic delays are directly learned in coordination with the neuronal adaptation. Furthermore, we showed that (1) it is possible to co-learn synaptic weights, delays, and neuronal adaptation parameters at the same time and (2) co-learning these parameters proved to mutually benefit the optimization of all learned parameters as shown for three speech recognition datasets.</p>
<p>We highlighted that this co-optimization leads to state-of-the-art performance in SNN on all investigated datasets. The superior performance can be attributed to two additional features: (1) Training the synaptic delays enables a neuron in the SNN to explicitly correlate temporally distanced features and (2) The trained neuronal adaptation allows a greater variety in spike patterns, widening the feature space to be explored.</p>
<p>We showed that for a very simple architecture, a 2-hidden layer is fully connected to feedforward network; we are able to compete against and even outperform larger ANN models, with a limited number of trainable SNN parameters. The performance of the presented SNN model shows the promise of SNN research on tasks with rich temporal dynamics, and, in particular, research on biologically inspired extensions to existing SNN models.</p>
<p>When comparing with larger ANN models, the performance of the presented SNN model is lacking. A future step in our research is therefore to investigate how learning delays and adaptation parameters are influenced by the model architecture. More advanced architectures such as convolutional spiking neural networks or experimenting with the training recurrent synapses could further bridge the gap with ANNs. Another exciting avenue to be explored is the chosen neuron model. As the proposed cAdLIF model is a particular generalized integrate-and-fire model version (Gerstner and Kistler, <xref ref-type="bibr" rid="B14">2002</xref>; Gerstner et al., <xref ref-type="bibr" rid="B15">2014</xref>) and many more exist, there are various neuron model extensions available to be investigated. Future research will point out that to what extent, these will be useful for the application in SNN in the context of (1) their additional performance in terms of classification accuracy and (2) the additional complexity for their deployment on dedicated neuromorphic hardware.</p>
<p>Another point to explore is that our study includes a fixed maximal delay, which needs to be defined before training and requires fine-tuning. Including adjustable delay caps could benefit our approach. In future studies, we intend to investigate the effect of co-learning the synaptic weights and delays on more complex neuron models and model architectures and validate them on non-sound datasets.</p></sec>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p></sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>LD: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. LV: Conceptualization, Methodology, Writing &#x02013; review &#x00026; editing. WV: Project administration, Supervision, Writing &#x02013; review &#x00026; editing. IT: Conceptualization, Supervision, Validation, Writing &#x02013; review &#x00026; editing. SL: Funding acquisition, Project administration, Resources, Supervision, Writing &#x02013; review &#x00026; editing.</p></sec>
</body>
<back>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This study was supported by a SB grant (1S87022N) from the Research Foundation Flanders (FWO).</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bittar</surname> <given-names>A.</given-names></name> <name><surname>Garner</surname> <given-names>P. N.</given-names></name></person-group> (<year>2022</year>). <article-title>A surrogate gradient spiking baseline for speech command recognition</article-title>. <source>Front. Neurosci</source>. <volume>16</volume>:<fpage>865897</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2022.865897</pub-id><pub-id pub-id-type="pmid">36117617</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brunel</surname> <given-names>N.</given-names></name> <name><surname>Hakim</surname> <given-names>V.</given-names></name> <name><surname>Richardson</surname> <given-names>M. J.</given-names></name></person-group> (<year>2003</year>). <article-title>Firing-rate resonance in a generalized integrate-and-fire neuron with subthreshold resonance</article-title>. <source>Phys. Rev. E</source> <volume>67</volume>:<fpage>051916</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevE.67.051916</pub-id><pub-id pub-id-type="pmid">12786187</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Bu</surname> <given-names>T.</given-names></name> <name><surname>Fang</surname> <given-names>W.</given-names></name> <name><surname>Ding</surname> <given-names>J.</given-names></name> <name><surname>Dai</surname> <given-names>P.</given-names></name> <name><surname>Yu</surname> <given-names>Z.</given-names></name> <name><surname>Huang</surname> <given-names>T.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Optimal ANN-SNN conversion for high-accuracy and ultra-low-latency spiking neural networks,&#x0201D;</article-title> in <source>The Tenth International Conference on Learning Representations</source>. <ext-link ext-link-type="uri" xlink:href="https://OpenReview.net">OpenReview.net</ext-link>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chakraborty</surname> <given-names>B.</given-names></name> <name><surname>Mukhopadhyay</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>Heterogeneous recurrent spiking neural network for spatio-temporal classification</article-title>. <source>Front. Neurosci</source>. <volume>17</volume>:<fpage>994517</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2023.994517</pub-id><pub-id pub-id-type="pmid">36793542</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cramer</surname> <given-names>B.</given-names></name> <name><surname>Stradmann</surname> <given-names>Y.</given-names></name> <name><surname>Schemmel</surname> <given-names>J.</given-names></name> <name><surname>Zenke</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>The heidelberg spiking data sets for the systematic evaluation of spiking neural networks</article-title>. <source>IEEE Transact. Neural Netw. Learn. Syst</source>. <volume>33</volume>, <fpage>2744</fpage>&#x02013;<lpage>2757</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2020.3044364</pub-id><pub-id pub-id-type="pmid">33378266</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>DeBole</surname> <given-names>M. V.</given-names></name> <name><surname>Taba</surname> <given-names>B.</given-names></name> <name><surname>Amir</surname> <given-names>A.</given-names></name> <name><surname>Akopyan</surname> <given-names>F.</given-names></name> <name><surname>Andreopoulos</surname> <given-names>A.</given-names></name> <name><surname>Risk</surname> <given-names>W. P.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Truenorth: accelerating from zero to 64 million neurons in 10 years</article-title>. <source>Computer</source> <volume>52</volume>, <fpage>20</fpage>&#x02013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.1109/MC.2019.2903009</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deckers</surname> <given-names>L.</given-names></name> <name><surname>Tsang</surname> <given-names>I. J.</given-names></name> <name><surname>Van Leekwijck</surname> <given-names>W.</given-names></name> <name><surname>Latr&#x000E9;</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>Extended liquid state machines for speech recognition</article-title>. <source>Front. Neurosci</source>. <volume>16</volume>:<fpage>1023470</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2022.1023470</pub-id><pub-id pub-id-type="pmid">36389242</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>S.</given-names></name> <name><surname>Gu</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Optimal conversion of conventional artificial neural networks to spiking neural networks,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source>. <ext-link ext-link-type="uri" xlink:href="https://OpenReview.net">OpenReview.net</ext-link>.</citation>
</ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Falez</surname> <given-names>P.</given-names></name> <name><surname>Tirilly</surname> <given-names>P.</given-names></name> <name><surname>Bilasco</surname> <given-names>I. M.</given-names></name> <name><surname>Devienne</surname> <given-names>P.</given-names></name> <name><surname>Boulet</surname> <given-names>P.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Multi-layered spiking neural network with target timestamp threshold adaptation and STDP,&#x0201D;</article-title> in <source>2019 International Joint Conference on Neural Networks (IJCNN)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1109/IJCNN.2019.8852346</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fang</surname> <given-names>W.</given-names></name> <name><surname>Yu</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Huang</surname> <given-names>T.</given-names></name> <name><surname>Masquelier</surname> <given-names>T.</given-names></name> <name><surname>Tian</surname> <given-names>Y.</given-names></name></person-group> (<year>2021a</year>). <article-title>Deep residual learning in spiking neural networks</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>34</volume>, <fpage>21056</fpage>&#x02013;<lpage>21069</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fang</surname> <given-names>W.</given-names></name> <name><surname>Yu</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Masquelier</surname> <given-names>T.</given-names></name> <name><surname>Huang</surname> <given-names>T.</given-names></name> <name><surname>Tian</surname> <given-names>Y.</given-names></name></person-group> (<year>2021b</year>). <article-title>&#x0201C;Incorporating learnable membrane time constant to enhance learning of spiking neural networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)</source>, <fpage>2661</fpage>&#x02013;<lpage>2671</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Garc&#x000ED;a-Mart&#x000ED;n</surname> <given-names>E.</given-names></name> <name><surname>Rodrigues</surname> <given-names>C. F.</given-names></name> <name><surname>Riley</surname> <given-names>G.</given-names></name> <name><surname>Grahn</surname> <given-names>H.</given-names></name></person-group> (<year>2019</year>). <article-title>Estimation of energy consumption in machine learning</article-title>. <source>J. Parallel Distrib. Comput</source>. <volume>134</volume>, <fpage>75</fpage>&#x02013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1016/j.jpdc.2019.07.007</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gast</surname> <given-names>R.</given-names></name> <name><surname>Schmidt</surname> <given-names>H.</given-names></name> <name><surname>Kn&#x00027;&#x00301;osche</surname> <given-names>T. R.</given-names></name></person-group> (<year>2020</year>). <article-title>A mean-field description of bursting dynamics in spiking neural networks with short-term adaptation</article-title>. <source>Neural Comput</source>. <volume>32</volume>, <fpage>1615</fpage>&#x02013;<lpage>1634</lpage>. <pub-id pub-id-type="doi">10.1162/neco_a_01300</pub-id><pub-id pub-id-type="pmid">32687770</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gerstner</surname> <given-names>W.</given-names></name> <name><surname>Kistler</surname> <given-names>W. M.</given-names></name></person-group> (<year>2002</year>). <source>Spiking Neuron Models: Single Neurons, Populations, Plasticity</source>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gerstner</surname> <given-names>W.</given-names></name> <name><surname>Kistler</surname> <given-names>W. M.</given-names></name> <name><surname>Naud</surname> <given-names>R.</given-names></name> <name><surname>Paninski</surname> <given-names>L.</given-names></name></person-group> (<year>2014</year>). <source>Neuronal Dynamics: From Single Neurons to Networks and Models of Cognition</source>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gong</surname> <given-names>Y.</given-names></name> <name><surname>Chung</surname> <given-names>Y.-A.</given-names></name> <name><surname>Glass</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Ast: audio spectrogram transformer</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.21437/Interspeech.2021-698</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Grappolini</surname> <given-names>E.</given-names></name> <name><surname>Subramoney</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Beyond weights: deep learning in spiking neural networks with pure synaptic-delay training,&#x0201D;</article-title> in <source>Proceedings of the 2023 International Conference on Neuromorphic Systems</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grimaldi</surname> <given-names>A.</given-names></name> <name><surname>Perrinet</surname> <given-names>L. U.</given-names></name></person-group> (<year>2023</year>). <article-title>Learning heterogeneous delays in a layer of spiking neurons for fast motion detection</article-title>. <source>Biol. Cybern</source>. <volume>117</volume>, <fpage>373</fpage>&#x02013;<lpage>387</lpage>. <pub-id pub-id-type="doi">10.1007/s00422-023-00975-8</pub-id><pub-id pub-id-type="pmid">37695359</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hammouamri</surname> <given-names>I.</given-names></name> <name><surname>Khalfaoui-Hassani</surname> <given-names>I.</given-names></name> <name><surname>Masquelier</surname> <given-names>T.</given-names></name></person-group> (<year>2023</year>). <article-title>Learning delays in spiking neural networks using dilated convolutions with learnable spacings</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.2306.17670</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D. P.</given-names></name> <name><surname>Ba</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). <article-title>Adam: a method for stochastic optimization</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.1412.6980</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Guo</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Deng</surname> <given-names>S.</given-names></name> <name><surname>Hai</surname> <given-names>Y.</given-names></name> <name><surname>Gu</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Differentiable spike: Rethinking gradient-descent for training spiking neural networks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems, Vol. 34</source>, eds <person-group person-group-type="editor"><name><surname>Ranzato</surname> <given-names>M.</given-names></name> <name><surname>Beygelzimer</surname> <given-names>A.</given-names></name> <name><surname>Dauphin</surname> <given-names>Y.</given-names></name> <name><surname>Liang</surname> <given-names>P.</given-names></name> <name><surname>Vaughan</surname> <given-names>J. W.</given-names></name></person-group> (<publisher-name>Curran Associates, Inc.</publisher-name>), <fpage>23426</fpage>&#x02013;<lpage>23439</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>J.-W.</given-names></name> <name><surname>Faber</surname> <given-names>D. S.</given-names></name></person-group> (<year>2002</year>). <article-title>Modulation of synaptic delay during synaptic plasticity</article-title>. <source>Trends Neurosci</source>. <volume>25</volume>, <fpage>449</fpage>&#x02013;<lpage>455</lpage>. <pub-id pub-id-type="doi">10.1016/S0166-2236(02)02212-9</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maass</surname> <given-names>W.</given-names></name></person-group> (<year>1997</year>). <article-title>Networks of spiking neurons: the third generation of neural network models</article-title>. <source>Neur. Netw</source>. <volume>10</volume>, <fpage>1659</fpage>&#x02013;<lpage>1671</lpage>. <pub-id pub-id-type="doi">10.1016/S0893-6080(97)00011-7</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maass</surname> <given-names>W.</given-names></name> <name><surname>Schmitt</surname> <given-names>M.</given-names></name></person-group> (<year>1999</year>). <article-title>On the complexity of learning for spiking neurons with temporal coding</article-title>. <source>Inf. Comp</source>. <volume>153</volume>, <fpage>26</fpage>&#x02013;<lpage>46</lpage>. <pub-id pub-id-type="doi">10.1006/inco.1999.2806</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Neftci</surname> <given-names>E. O.</given-names></name> <name><surname>Mostafa</surname> <given-names>H.</given-names></name> <name><surname>Zenke</surname> <given-names>F.</given-names></name></person-group> (<year>2019</year>). <article-title>Surrogate gradient learning in spiking neural networks: Bringing the power of gradient-based optimization to spiking neural networks</article-title>. <source>IEEE Signal Process. Mag</source>. <volume>36</volume>, <fpage>51</fpage>&#x02013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.1109/MSP.2019.2931595</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Orchard</surname> <given-names>G.</given-names></name> <name><surname>Etienne-Cummings</surname> <given-names>R.</given-names></name></person-group> (<year>2014</year>). <article-title>Bioinspired visual motion estimation</article-title>. <source>Proc. IEEE</source> <volume>102</volume>, <fpage>1520</fpage>&#x02013;<lpage>1536</lpage>. <pub-id pub-id-type="doi">10.1109/JPROC.2014.2346763</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Orchard</surname> <given-names>G.</given-names></name> <name><surname>Frady</surname> <given-names>E. P.</given-names></name> <name><surname>Rubin</surname> <given-names>D. B. D.</given-names></name> <name><surname>Sanborn</surname> <given-names>S.</given-names></name> <name><surname>Shrestha</surname> <given-names>S. B.</given-names></name> <name><surname>Sommer</surname> <given-names>F. T.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Efficient neuromorphic signal processing with loihi 2,&#x0201D;</article-title> in <source>2021 IEEE Workshop on Signal Processing Systems (SiPS)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>254</fpage>&#x02013;<lpage>259</lpage>. <pub-id pub-id-type="doi">10.1109/SiPS52927.2021.00053</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Paszke</surname> <given-names>A.</given-names></name> <name><surname>Gross</surname> <given-names>S.</given-names></name> <name><surname>Massa</surname> <given-names>F.</given-names></name> <name><surname>Lerer</surname> <given-names>A.</given-names></name> <name><surname>Bradbury</surname> <given-names>J.</given-names></name> <name><surname>Chanan</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;PyTorch: an imperative style, high-performance deep learning library,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds <person-group person-group-type="editor"><name><surname>Wallach</surname> <given-names>H.</given-names></name> <name><surname>Larochelle</surname> <given-names>H.</given-names></name> <name><surname>Beygelzimer</surname> <given-names>A.</given-names></name> <name><surname>d&#x00027;Alch&#x000E9;-Buc</surname> <given-names>F.</given-names></name> <name><surname>Fox</surname> <given-names>E.</given-names></name> <name><surname>Garnett</surname> <given-names>R.</given-names></name></person-group> (Curran Associates, Inc.). Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2019/file/bdbca288fee7f92f2bfa9f7012727740-Paper.pdf">https://proceedings.neurips.cc/paper_files/paper/2019/file/bdbca288fee7f92f2bfa9f7012727740-Paper.pdf</ext-link></citation>
</ref>
<ref id="B29">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Pati&#x000F1;o-Saucedo</surname> <given-names>A.</given-names></name> <name><surname>Yousefzadeh</surname> <given-names>A.</given-names></name> <name><surname>Tang</surname> <given-names>G.</given-names></name> <name><surname>Corradi</surname> <given-names>F.</given-names></name> <name><surname>Linares-Barranco</surname> <given-names>B.</given-names></name> <name><surname>Sifalakis</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Empirical study on the efficiency of spiking neural networks with axonal delays, and algorithm-hardware benchmarking,&#x0201D;</article-title> in <source>2023 IEEE International Symposium on Circuits and Systems (ISCAS)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1</fpage>&#x02013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/ISCAS46773.2023.10181778</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Perez-Nieves</surname> <given-names>N.</given-names></name> <name><surname>Leung</surname> <given-names>V. C.</given-names></name> <name><surname>Dragotti</surname> <given-names>P. L.</given-names></name> <name><surname>Goodman</surname> <given-names>D. F.</given-names></name></person-group> (<year>2021</year>). <article-title>Neural heterogeneity promotes robust learning</article-title>. <source>Nat. Commun</source>. <volume>12</volume>:<fpage>5791</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-021-26022-3</pub-id><pub-id pub-id-type="pmid">34608134</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Purves</surname> <given-names>D.</given-names></name> <name><surname>Augustine</surname> <given-names>G.</given-names></name> <name><surname>Fitzpatrick</surname> <given-names>D.</given-names></name> <name><surname>Katz</surname> <given-names>L.</given-names></name> <name><surname>LaMantia</surname> <given-names>A.</given-names></name> <name><surname>McNamara</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2001</year>). <article-title>&#x0201C;The organization of the nervous system,&#x0201D;</article-title> in <source>Neuroscience</source>, eds <person-group person-group-type="editor"><name><surname>Purves</surname> <given-names>D.</given-names></name> <name><surname>Augustine</surname> <given-names>G. J.</given-names></name> <name><surname>Fitzpatrick</surname> <given-names>D.</given-names></name> <name><surname>Katz</surname> <given-names>L. C.</given-names></name> <name><surname>LaMantia</surname> <given-names>A. S.</given-names></name> <name><surname>McNamara</surname> <given-names>J. O.</given-names></name></person-group> (<publisher-loc>Sunderland, MA</publisher-loc>: <publisher-name>Sinauer Associates</publisher-name>).</citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rathi</surname> <given-names>N.</given-names></name> <name><surname>Roy</surname> <given-names>K.</given-names></name></person-group> (<year>2023</year>). <article-title>Diet-snn: a low-latency spiking neural network with direct input encoding and leakage and threshold optimization</article-title>. <source>IEEE Transact. Neural Netw. Learn. Syst</source>. <volume>34</volume>, <fpage>3174</fpage>&#x02013;<lpage>3182</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2021.3111897</pub-id><pub-id pub-id-type="pmid">34596559</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Salaj</surname> <given-names>D.</given-names></name> <name><surname>Subramoney</surname> <given-names>A.</given-names></name> <name><surname>Kraisnikovic</surname> <given-names>C.</given-names></name> <name><surname>Bellec</surname> <given-names>G.</given-names></name> <name><surname>Legenstein</surname> <given-names>R.</given-names></name> <name><surname>Maass</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>Spike frequency adaptation supports network computations on temporally dispersed information</article-title>. <source>Elife</source> <volume>10</volume>:<fpage>e65459</fpage>. <pub-id pub-id-type="doi">10.7554/eLife.65459</pub-id><pub-id pub-id-type="pmid">34310281</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Shrestha</surname> <given-names>S. B.</given-names></name> <name><surname>Orchard</surname> <given-names>G.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;SLAYER: spike layer error reassignment in time,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, eds <person-group person-group-type="editor"><name><surname>Bengio</surname> <given-names>S.</given-names></name> <name><surname>Wallach</surname> <given-names>H.</given-names></name> <name><surname>Larochelle</surname> <given-names>H.</given-names></name> <name><surname>Grauman</surname> <given-names>K.</given-names></name> <name><surname>Cesa-Bianchi</surname> <given-names>N.</given-names></name> <name><surname>Garnett</surname> <given-names>R.</given-names></name></person-group> (Curran Associates, Inc.). Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2018/file/82f2b308c3b01637c607ce05f52a2fed-Paper.pdf">https://proceedings.neurips.cc/paper_files/paper/2018/file/82f2b308c3b01637c607ce05f52a2fed-Paper.pdf</ext-link></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Srivastava</surname> <given-names>N.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name> <name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R.</given-names></name></person-group> (<year>2014</year>). <article-title>Dropout: a simple way to prevent neural networks from overfitting</article-title>. <source>J. Mach. Learn. Res</source>. <volume>15</volume>, <fpage>1929</fpage>&#x02013;<lpage>1958</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>St&#x000F6;ckl</surname> <given-names>C.</given-names></name> <name><surname>Maass</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>Optimized spiking neurons can classify images with high accuracy through temporal coding with two spikes</article-title>. <source>Nat. Mach. Intell</source>. <volume>3</volume>, <fpage>230</fpage>&#x02013;<lpage>238</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-021-00311-4</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>P.</given-names></name> <name><surname>Chua</surname> <given-names>Y.</given-names></name> <name><surname>Devos</surname> <given-names>P.</given-names></name> <name><surname>Botteldooren</surname> <given-names>D.</given-names></name></person-group> (<year>2023a</year>). <article-title>Learnable axonal delay in spiking neural networks improves spoken word recognition</article-title>. <source>Front. Neurosci</source>. <volume>17</volume>:<fpage>1275944</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2023.1275944</pub-id><pub-id pub-id-type="pmid">38027508</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>P.</given-names></name> <name><surname>Eqlimi</surname> <given-names>E.</given-names></name> <name><surname>Chua</surname> <given-names>Y.</given-names></name> <name><surname>Devos</surname> <given-names>P.</given-names></name> <name><surname>Botteldooren</surname> <given-names>D.</given-names></name></person-group> (<year>2023b</year>). <article-title>&#x0201C;Adaptive axonal delays in feedforward spiking neural networks for accurate spoken word recognition,&#x0201D;</article-title> in <source>ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1</fpage>&#x02013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>P.</given-names></name> <name><surname>Zhu</surname> <given-names>L.</given-names></name> <name><surname>Botteldooren</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Axonal delay as a short-term memory for feed forward deep spiking neural networks,&#x0201D;</article-title> in <source>ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>8932</fpage>&#x02013;<lpage>8936</lpage>. <pub-id pub-id-type="doi">10.1109/ICASSP43922.2022.9747411</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>R.</given-names></name> <name><surname>Cohen</surname> <given-names>G.</given-names></name> <name><surname>Stiefel</surname> <given-names>K. M.</given-names></name> <name><surname>Hamilton</surname> <given-names>T. J.</given-names></name> <name><surname>Tapson</surname> <given-names>J.</given-names></name> <name><surname>van Schaik</surname> <given-names>A.</given-names></name></person-group> (<year>2013</year>). <article-title>An fpga implementation of a polychronous spiking neural network with delay adaptation</article-title>. <source>Front. Neurosci</source>. <volume>7</volume>:<fpage>14</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2013.00014</pub-id><pub-id pub-id-type="pmid">23408739</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Lin</surname> <given-names>X.</given-names></name> <name><surname>Dang</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). <article-title>A delay learning algorithm based on spike train kernels for spiking neurons</article-title>. <source>Front. Neurosci</source>. <volume>13</volume>:<fpage>252</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2019.00252</pub-id><pub-id pub-id-type="pmid">30971877</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Warden</surname> <given-names>P.</given-names></name></person-group> (<year>2018</year>). <article-title>Speech commands: a dataset for limited-vocabulary speech recognition</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.1804.03209</pub-id></citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yao</surname> <given-names>M.</given-names></name> <name><surname>Zhao</surname> <given-names>G.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>Y.</given-names></name> <name><surname>Deng</surname> <given-names>L.</given-names></name> <name><surname>Tian</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Attention spiking neural networks</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>45</volume>, <fpage>93939410</fpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2023.3241201</pub-id></citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yao</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>F.</given-names></name> <name><surname>Mo</surname> <given-names>Z.</given-names></name> <name><surname>Cheng</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Glif: a unified gated leaky integrate-and-fire neuron for spiking neural networks</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>35</volume>, <fpage>32160</fpage>&#x02013;<lpage>32171</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>B.</given-names></name> <name><surname>Corradi</surname> <given-names>F.</given-names></name> <name><surname>Boht&#x000E9;</surname> <given-names>S. M.</given-names></name></person-group> (<year>2021</year>). <article-title>Accurate and efficient time-domain classification with adaptive spiking recurrent neural networks</article-title>. <source>Nat. Mach. Intell</source>. <volume>3</volume>, <fpage>905</fpage>&#x02013;<lpage>913</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-021-00397-w</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yousefzadeh</surname> <given-names>A.</given-names></name> <name><surname>Van Schaik</surname> <given-names>G.-J.</given-names></name> <name><surname>Tahghighi</surname> <given-names>M.</given-names></name> <name><surname>Detterer</surname> <given-names>P.</given-names></name> <name><surname>Traferro</surname> <given-names>S.</given-names></name> <name><surname>Hijdra</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;Seneca: scalable energy-efficient neuromorphic computer architecture,&#x0201D;</article-title> in <source>2022 IEEE 4th International Conference on Artificial Intelligence Circuits and Systems (AICAS)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>371</fpage>&#x02013;<lpage>374</lpage>. <pub-id pub-id-type="doi">10.1109/AICAS54282.2022.9870025</pub-id></citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zenke</surname> <given-names>F.</given-names></name> <name><surname>Vogels</surname> <given-names>T. P.</given-names></name></person-group> (<year>2021</year>). <article-title>The remarkable robustness of surrogate gradient learning for instilling complex function in spiking neural networks</article-title>. <source>Neural Comput</source>. <volume>33</volume>, <fpage>899</fpage>&#x02013;<lpage>925</lpage>. <pub-id pub-id-type="doi">10.1162/neco_a_01367</pub-id><pub-id pub-id-type="pmid">33513328</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Wu</surname> <given-names>J.</given-names></name> <name><surname>Belatreche</surname> <given-names>A.</given-names></name> <name><surname>Pan</surname> <given-names>Z.</given-names></name> <name><surname>Xie</surname> <given-names>X.</given-names></name> <name><surname>Chua</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Supervised learning in spiking neural networks with synaptic delay-weight plasticity</article-title>. <source>Neurocomputing</source> <volume>409</volume>, <fpage>103</fpage>&#x02013;<lpage>118</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2020.03.079</pub-id></citation>
</ref>
<ref id="B49">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Z.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>He</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>YAN</surname> <given-names>S.</given-names></name> <name><surname>Tian</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>&#x0201C;Spikformer: when spiking neural network meets transformer,&#x0201D;</article-title> in <source>The Eleventh International Conference on Learning Representations</source>. <ext-link ext-link-type="uri" xlink:href="https://OpenReview.net">OpenReview.net</ext-link>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>R.-J.</given-names></name> <name><surname>Zhao</surname> <given-names>Q.</given-names></name> <name><surname>Eshraghian</surname> <given-names>J. K.</given-names></name></person-group> (<year>2023</year>). <article-title>Spikegpt: generative pre-trained language model with spiking neural networks</article-title>. <source>arXiv</source> [preprint]. <pub-id pub-id-type="doi">10.48550/arXiv.2302.13939</pub-id></citation>
</ref>
</ref-list>
</back>
</article> 