<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2025.1665778</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Balancing accuracy and efficiency: co-design of hybrid quantization and unified computing architecture for spiking neural networks</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Jiahao</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3128181/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Ming</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dong</surname>
<given-names>Heng</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lan</surname>
<given-names>Bin</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Yuxin</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chen</surname>
<given-names>He</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhuang</surname>
<given-names>Yin</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2185984/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Xie</surname>
<given-names>Yizhuang</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Liang</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2101027/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>National Key Laboratory of Space-Born Intelligent Information Processing, Beijing Institute of Technology</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Sichuan Tianfu New Area, Beijing Institute of Technology Innovation Equipment Research Institute</institution>, <addr-line>Chengdu</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Beijing Institute of Technology Chongqing Innovation Center</institution>, <addr-line>Chongqing</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1596021/overview">Lan Du</ext-link>, Monash University, Australia</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2402988/overview">Dhruva Ghai</ext-link>, Oriental University, India</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2370080/overview">Ruokai Yin</ext-link>, Yale University, United States</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: He Chen, <email>chenhe@bit.edu.cn</email></corresp>
<corresp id="c002">Yizhuang Xie, <email>xyz551_bit@bit.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>19</volume>
<elocation-id>1665778</elocation-id>
<history>
<date date-type="received">
<day>14</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Li, Xu, Dong, Lan, Liu, Chen, Zhuang, Xie and Chen.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Li, Xu, Dong, Lan, Liu, Chen, Zhuang, Xie and Chen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The deployment of Spiking Neural Networks (SNNs) on resource-constrained edge devices is hindered by a critical algorithm-hardware mismatch: a fundamental trade-off between the accuracy degradation caused by aggressive quantization and the resource redundancy stemming from traditional decoupled hardware designs. To bridge this gap, we present a novel algorithm-hardware co-design framework centered on a Ternary-8-bit Hybrid Weight Quantization (T8HWQ) scheme. Our approach recasts SNN computation into a unified &#x201C;8-bit &#x00D7; 2-bit&#x201D; paradigm by quantizing first-layer weights to 2 bits and subsequent layers to 8 bits. This standardization directly enables the design of a unified PE architecture, eliminating the resource redundancy inherent in decoupled designs. To mitigate the accuracy degradation caused by aggressive first-layer quantization, we first propose a channel-wise dual compensation strategy. This method synergizes channel-wise quantization optimization with adaptive threshold neurons, leveraging reparameterization techniques to restore model accuracy without incurring additional inference overhead. Building upon T8HWQ, we propose a novel unified computing architecture that overcomes the inefficiencies of traditional decoupled designs by efficiently multiplexing processing arrays. Experimental results support our approach: On CIFAR-100, our method achieves near-lossless accuracy (&#x003C;0.7% degradation vs. full precision) with a single time step, matching state-of-the-art low-bit SNNs. At the hardware level, implementation results on the Xilinx Virtex 7 platform demonstrate that our unified computing unit conserves 20.2% of lookup table (LUT) resources compared to traditional decoupled architectures. This work delivers a 6&#x202F;&#x00D7;&#x202F;throughput improvement over state-of-the-art SNN accelerators&#x2014;with comparable resource utilization and lower power consumption. Our integrated solution thus advances the practical implementation of high-performance, low-latency SNNs on resource-constrained edge devices.</p>
</abstract>
<kwd-group>
<kwd>spiking neural networks</kwd>
<kwd>quantization</kwd>
<kwd>field-programmable gate array</kwd>
<kwd>algorithm-hardware co-design</kwd>
<kwd>unified processing elements</kwd>
<kwd>resource-constrained devices</kwd>
</kwd-group>
<counts>
<fig-count count="11"/>
<table-count count="9"/>
<equation-count count="20"/>
<ref-count count="44"/>
<page-count count="16"/>
<word-count count="11818"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Neuromorphic Engineering</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>A fundamental tension exists between the escalating computational demands of sophisticated artificial neural networks (ANNs) and the stringent computing, storage, and power constraints inherent to edge devices (<xref ref-type="bibr" rid="ref16">Hinton et al., 2015</xref>; <xref ref-type="bibr" rid="ref22">Li et al., 2017</xref>; <xref ref-type="bibr" rid="ref39">Wei et al., 2024</xref>). Although ANNs demonstrate exceptional performance across diverse computational tasks, their reliance on extensive model sizes and dense multiply-accumulate (MAC) operations inherently leads to prohibitively high energy consumption and significant computational latency. As a promising solution, brain-inspired Spiking Neural Networks (SNNs) encode and transmit information through sparse spiking signals (<xref ref-type="bibr" rid="ref26">Maass, 1997</xref>), enabling hardware-level computational sparsity (<xref ref-type="bibr" rid="ref30">Plagwitz et al., 2023</xref>). When implemented on reconfigurable platforms such as field-programmable gate arrays (FPGAs), this sparsity inherently bypasses redundant operations, drastically reducing dynamic power consumption and enabling high-performance edge AI systems (<xref ref-type="bibr" rid="ref18">Karamimanesh et al., 2025</xref>).</p>
<p>The practical deployment of SNNs is fundamentally at odds with critical performance and hardware limitations. A primary impediment is the persistent accuracy gap. SNN training algorithms, while advancing, have yet to consistently match the performance of structurally equivalent ANNs (<xref ref-type="bibr" rid="ref25">Luo et al., 2025</xref>; <xref ref-type="bibr" rid="ref35">Su et al., 2023</xref>). Furthermore, unlocking the profound energy efficiency of SNNs is contingent upon a transition from inefficient von Neumann-based simulations to specialized hardware, such as FPGAs and neuromorphic chips (<xref ref-type="bibr" rid="ref18">Karamimanesh et al., 2025</xref>). In this context, model quantization represents a pivotal strategy (<xref ref-type="bibr" rid="ref17">Jacob et al., 2018</xref>; <xref ref-type="bibr" rid="ref8">Courbariaux et al., 2015</xref>). By reducing the bit width of model weights, quantization can drastically curtails storage requirements and computational complexity, a crucial step for adapting SNNs to these resource-constrained hardware platforms.</p>
<p>Current research in SNN quantization is bifurcated into two distinct trajectories, yet both converge on a significant, unresolved hardware implementation challenge. The first path involves moderate quantization to 8-bit or 4-bit precision, a strategy that achieves effective model compression while maintaining performance, but offers limited optimization for ultra-resource-constrained environments (<xref ref-type="bibr" rid="ref45">Zou et al., 2024</xref>; <xref ref-type="bibr" rid="ref31">Putra and Shafique, 2021</xref>; <xref ref-type="bibr" rid="ref7">Chowdhury et al., 2021</xref>). The second, more aggressive approach utilizes low-bit quantization, such as binary (1-bit) (<xref ref-type="bibr" rid="ref4">Cao et al., 2025</xref>; <xref ref-type="bibr" rid="ref10">Eshraghian and Lu, 2022</xref>) or ternary (~2-bit) schemes (<xref ref-type="bibr" rid="ref13">Hasssan et al., 2024</xref>). This method dramatically reduces hardware complexity by converting multiplications into efficient bitwise operations or additions, though it frequently incurs severe accuracy degradation (<xref ref-type="bibr" rid="ref45">Zou et al., 2024</xref>).</p>
<p>A critical flaw emerges at the hardware level, where deployment of these multi-precision networks is often architecturally &#x201C;decoupled&#x201D;. As illustrated in <xref ref-type="fig" rid="fig1">Figure 1</xref>, this paradigm necessitates designing separated processing elements (PEs) for different data bit widths (e.g., 8-bit and 2-bit). Such a decoupled PE design is fundamentally inefficient, failing to achieve full utilization of valuable on-chip computational resources.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Comparison of unified PEs with traditional methods based on SNNs.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Comparison of traditional and proposed methods for processing layers. The traditional method shows an 8-bit input to PE1 in the first layer and 2-bit features in other layers, with 8-bit weights. The proposed method uses 8-bit input and 2-bit weights in the first layer to PE2'. Other layers process 2-bit features with 8-bit weights to PE2'. Both methods output 2-bit values ranging from -1 to 1.</alt-text>
</graphic>
</fig>
<p>A significant algorithm-hardware gap currently impedes the advancement of SNN deployment, presenting distinct yet interrelated challenges. At the hardware level, contemporary FPGA architectures lack unified PEs capable of supporting mixed-precision computation, thereby failing to maximize resource efficiency and computational throughput. Concurrently, at the algorithmic level, a critical need exists for hardware-aware, low-bit quantization methods that preserve model accuracy without introducing computational complexity.</p>
<p>This paper directly confronts this algorithm-hardware divide by proposing a novel algorithm-hardware co-design for optimizing FPGA-based computing units. Our central insight stems from identifying two distinct operational modes during SNN inference on FPGAs: (1) initial-layer computations involving weight and input pixel data, and (2) subsequent-layer computations involving weight and spike features. To reconcile these modes into a single and efficient architecture, we introduce the T8HWQ scheme. This strategy implements ternary (~2-bit) quantization for the first-layer weights and 8-bit quantization for subsequent layers. Consequently, first-layer operations are standardized as 2-bit weight &#x00D7; 8-bit input multiplications, while subsequent-layer operations become 8-bit weight &#x00D7; 2-bit spike interactions. This innovative approach unifies all network computations into a consistent 8-bit &#x00D7; 2-bit paradigm, enabling the design of a highly resource-efficient, unified computing architecture.</p>
<p>However, the aggressive first-layer quantization inevitably degrades model performance. To address this, we propose a channel-wise dual compensation strategy that recovers accuracy without increasing inference-stage computation costs. Simultaneously, building on the T8HWQ scheme, we design a unified FPGA computing architecture that supports all target operations while maximizing hardware resource reuse. In summary, the contribution of this article is as follows:</p>
<list list-type="order">
<list-item>
<p>We propose the first algorithm-hardware co-designed T8HWQ scheme to address the heterogeneous computational patterns in SNNs. Our approach reconciles first-layer &#x201C;weight &#x00D7; 8-bit image pixel&#x201D; operations and subsequent-layer &#x201C;weight &#x00D7; 2-bit spike&#x201D; operations into a uniform 8-bit &#x00D7; 2-bit computation. This innovative strategy preserves model accuracy and provides a robust algorithmic basis for designing efficient, unified FPGA computing architectures.</p>
</list-item>
<list-item>
<p>To overcome the performance deficit from aggressive quantization, we introduce a novel compensation strategy with zero computational overhead at inference. This is achieved through two synergistic mechanisms: channel-wise quantization to account for feature variations and a channel-wise adaptive threshold neuron to dynamically regulate spike activation. Both are seamlessly integrated into the network weights via reparameterization, enabling our model to achieve accuracy on par with full-precision counterparts.</p>
</list-item>
<list-item>
<p>In this paper, a unified computing architecture based on the T8HWQ scheme is designed and implemented on the FPGA. By multiplexing the PE computing array, this unified approach eliminates the resource redundancy inherent in traditional decoupled PEs, achieving optimal hardware utilization without compromising computational throughput.</p>
</list-item>
<list-item>
<p>We evaluate our proposed method through both algorithm and hardware experiments. Algorithmically, on the CIFAR-10 and CIFAR-100 datasets, our approach attain near-lossless accuracy (&#x003C;0.7% degradation) relative to the full-precision model in a single time step, performing competitively with other state-of-the-art (SOTA) low-bit SNNs. On the hardware front, implementation on a Xilinx XC7VX690T platform confirm that the unified computing architecture reduces lookup table (LUT) resource utilization by 20.2% compared to the traditional architecture. Compared with other advanced SNN hardware accelerators, our design delivers 6&#x202F;&#x00D7;&#x202F;greater throughput than advanced SNN hardware accelerators at a comparable resource and power budget.</p>
</list-item>
</list>
</sec>
<sec id="sec2">
<label>2</label>
<title>Related works</title>
<p>To deploy lightweight SNNs, quantization serves as a critical approach for compressing these networks. Unlike traditional ANNs, which utilize real-valued activations, SNNs communicate through spikes, effectively reducing the storage requirements for feature maps.</p>
<p>To further reduce storage space, research on quantization has focused on minimizing the bit-width of weights. All parameters are quantized to integers, including membrane potential and firing threshold (<xref ref-type="bibr" rid="ref45">Zou et al., 2024</xref>). Furthermore, the Q-SpiNN framework is proposed for quantizing SNNs by addressing different parameters, precision levels, and rounding schemes (<xref ref-type="bibr" rid="ref31">Putra and Shafique, 2021</xref>), reducing the bit-width to 5 bits. Additionally, spatial and temporal pruning of SNNs are implemented, decreasing the bit-width to 5 bits (<xref ref-type="bibr" rid="ref7">Chowdhury et al., 2021</xref>). Moreover, quantization-aware training (QAT) with stacked gradient surrogation is proposed for integer-only SNNs, reducing the bit-width to 4 bits (<xref ref-type="bibr" rid="ref13">Hasssan et al., 2024</xref>).</p>
<p>In recent years, to further reduce the latency of SNNs and achieve more lightweight models, researchers have further decreased the bit-width from 2 bits to 1 bit while maintaining a high latency (exceeding 5). SQUAT is proposed to enhance performance by achieving 2-bit weigh while using 25 time steps (<xref ref-type="bibr" rid="ref36">Venkatesh et al., 2024</xref>). MINT quantizes both weights and membrane potentials to extremely low precisions in 2 bits with 8 time steps (<xref ref-type="bibr" rid="ref41">Yin et al., 2024</xref>). Furthermore, researchers compress the weights to 1 bit (<xref ref-type="bibr" rid="ref4">Cao et al., 2025</xref>; <xref ref-type="bibr" rid="ref10">Eshraghian and Lu, 2022</xref>). <xref ref-type="bibr" rid="ref9">Deng et al. (2023)</xref> propose connection pruning and weight quantization methods using ADMM optimization and activity regularization, successfully reducing the bit-width to 1, while the time steps are limited to 10 (<xref ref-type="bibr" rid="ref9">Deng et al., 2023</xref>). <xref ref-type="bibr" rid="ref33">Shymyrbay et al. (2023)</xref> present a framework for quantizing SNN models using a differentiable quantization function based on a linear combination of sigmoid functions achieving 1-bit weight while using 10 time steps.</p>
<p>Despite the significant reduction in storage space achieved by decreasing the bit width to 1-bit, performance suffers greatly. Furthermore, existing quantization studies primarily focus on minimizing bit width from the perspective of individual algorithms, overlooking the hardware costs associated with network image encoding. Given that images are typically represented in 8 bits, inconsistencies may arise between the computations of the first layer and those of subsequent layers, resulting in low resource reuse efficiency in FPGAs, as illustrated in <xref ref-type="fig" rid="fig1">Figure 1</xref>.</p>
<p>In summary, the current quantization techniques for SNNs exhibit three main characteristics: First, the use of a uniform bit width for weight quantization across all layers leads to low FPGA resource reuse efficiency, resulting in wasted computational resources. Second, as the weight bit width decreases to lower values (e.g., 2 bits), network performance declines severely, failing to achieve an optimal balance between accuracy and compression. Finally, existing quantization methods primarily focus on multiple time steps, resulting in high latency for SNNs. Consequently, these methods are unsuitable for resource-constrained and real-time applications.</p>
<p>To save FPGA computational resources, enhance resource efficiency, and maintain high performance under low-latency SNN conditions, we employ a ternary spike neuron with stronger information representation capabilities for SNNs. This research focuses on SNN quantization from the perspective of hardware-software co-design for FPGA implementations.</p>
</sec>
<sec sec-type="methods" id="sec3">
<label>3</label>
<title>Methodology</title>
<sec id="sec4">
<label>3.1</label>
<title>The overall co-design of the quantization algorithm and hardware</title>
<p>This paper adopts the algorithm-hardware co-design paradigm, and at the algorithm level, the core lies in designing a quantitative strategy that is tailored to the features of the hardware. At the algorithm level, the T8HWQ quantization method is proposed to balance resource usage, processing time, and model performance. At the hardware level, FPGA design emphasizes the efficient utilization of hardware resources. Through the close collaboration of these two aspects, the aim is to build a resource-saving SNN system with both high-performance computing capabilities and low latency.</p>
<p>The algorithm-hardware co-design paradigm presented in this study includes quantization strategies at the algorithmic level, as well as design and analysis related to FPGA design, as illustrated in the <xref ref-type="fig" rid="fig2">Figure 2</xref>.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Algorithm-hardware co-design framework based on SNNs.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g002.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Framework diagram illustrating an algorithm and hardware co-design based on ternary spike neurons. The algorithm involves ternary-eight-bit hybrid weight quantization, featuring a dual compensation quantization strategy within a QAT framework. The hardware design uses a unified processing element implemented on FPGA. The diagram includes components like convolutional layers, quantization functions, and ternary spike neurons with channel-wise and layer-wise quantization strategies, showing data flow from eight-bit inputs to two-bit outputs.</alt-text>
</graphic>
</fig>
<sec id="sec5">
<label>3.1.1</label>
<title>Ternary-8bit hybrid weight quantization</title>
<p>At the algorithmic level, we have developed a hybrid bit-width quantization method T8HWQ. This method quantizes the weights of the first layer of the network into a ternary set {&#x2212;1, 0, 1}. This design offers dual advantages: first, the ternary weights align with the discrete nature of neuron spike events in SNNs, allowing traditional multiplication operations to be converted into more efficient addition operations; second, ternary quantization compresses the weight storage to 2 bits, thereby reducing storage requirements. For the other layers, the data is quantized to 8 bits. Compared to traditional quantization methods, this hybrid bit-width design couples the computations across layers and simplifies the hardware design.</p>
</sec>
<sec id="sec6">
<label>3.1.2</label>
<title>The channel-wise dual compensation strategy based on QAT framework</title>
<p>Since the first layer serves as encoding, quantizing its weights to 2 bits inevitably leads to a certain degree of performance degradation. To address this issue, we have designed a channel-wise dual compensation strategy. This strategy effectively reduces the coding loss caused by the low bit-width quantization of the first layer by sequentially applying channel-wise quantization compensation (<xref ref-type="bibr" rid="ref19">Krishnamoorthi, 2018</xref>) and channel-wise adaptive threshold ternary neurons compensation. Furthermore, we have constructed a quantization framework based on QAT that continuously adjusts the model to its optimal compensated state during the training phase, thereby enhancing the overall performance of the network.</p>
</sec>
<sec id="sec7">
<label>3.1.3</label>
<title>Design and implementation of a unified PE based on FPGA</title>
<p>From the perspective of FPGA design, the aforementioned hybrid bit-width quantization strategy offers significant advantages. In traditional hardware designs, processing data of different bit-widths typically requires distinct computational modules. However, in our design, the quantization strategy allows the FPGA to primarily handle 2-bit and 8-bit data. Based on this characteristic, we have developed a unified PE for ternary spiking neurons using FPGA. This unified PE efficiently processes both bit-widths, greatly simplifying the hardware architecture and thus saving FPGA hardware resources.</p>
</sec>
</sec>
<sec id="sec8">
<label>3.2</label>
<title>The channel-wise dual compensation strategy based on QAT framework</title>
<p>In this section, we propose a channel-wise dual compensation strategy based on QAT. We innovatively introduce adaptive threshold ternary spike neurons to channel-wise quantization, and the combination of these two compensation methods is termed the dual compensation strategy.</p>
<sec id="sec9">
<label>3.2.1</label>
<title>Channel-wise adaptive threshold ternary spike neuron</title>
<p>LIF (Leaky Integrate-and-Fire) neurons are widely used as mathematical models of neuronal activity, including spike firing and the update of membrane potential. This paper adopts the ternary spike neuron (<xref ref-type="bibr" rid="ref12">Guo et al., 2024</xref>). Compared to binary spikes, ternary spikes convey richer information. By making the positive and negative threshold parameters learnable, neurons can adaptively adjust their activation based on data and tasks, thereby enhancing overall network performance. The proposed neuron model can be represented by <xref ref-type="disp-formula" rid="EQ1">Equations (1)</xref> and <xref ref-type="disp-formula" rid="EQ2">(2)</xref>:</p>
<disp-formula id="EQ1">
<label>(1)</label>
<mml:math id="M1">
<mml:msubsup>
<mml:mi>U</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>&#x03C4;</mml:mi>
<mml:msubsup>
<mml:mi>U</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>W</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>O</mml:mi>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</disp-formula>
<disp-formula id="EQ2">
<label>(2)</label>
<mml:math id="M2">
<mml:msubsup>
<mml:mi>O</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mtext>if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi>U</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x2265;</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>+</mml:mo>
</mml:msubsup>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="0.25em"/>
<mml:mtext>otherwise</mml:mtext>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="1em"/>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mtext>if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi>U</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x2264;</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msubsup>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>where <italic>&#x03C4;</italic> is a constant that describes membrane potential decaying. <italic>U<sub>c</sub><sup>l</sup></italic>(<italic>t</italic>) represents the membrane potential at time step <italic>t</italic> in the <italic>c</italic>-th channel of the <italic>l</italic>-th layer. <italic>I</italic> is the presynaptic input and <italic>W<sub>c</sub><sup>l</sup>O<sub>c</sub></italic><sup><italic>l</italic>-1</sup>(<italic>t</italic>) is the accumulation of spikes from the neurons of layer <italic>l</italic>-1. <italic>W</italic> is the weight of the neuron. <italic>O</italic> is the spiking output of the neuron from the previous layer. <italic>V<sub>c</sub></italic><sup>+</sup>&#x202F;&#x003E;&#x202F;0 is the positive threshold of the <italic>c</italic>-th channel, <italic>V<sub>c</sub></italic><sup>&#x2212;</sup>&#x202F;&#x003C;&#x202F;0 is the negative threshold of the <italic>c</italic>-th channel. These two learnable thresholds enable neurons to find appropriate activation thresholds for different channels. We employ a soft reset mechanism represented by <xref ref-type="disp-formula" rid="EQ3">Equation (3)</xref>:</p>
<disp-formula id="EQ3">
<label>(3)</label>
<mml:math id="M3">
<mml:msubsup>
<mml:mi>U</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi>U</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>+</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi>U</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi>U</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mtext mathvariant="italic">if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi>O</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mtext mathvariant="italic">if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi>O</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mtext mathvariant="italic">if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi>O</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>SNNs output spike sequences instead of continuous numerical values, presenting challenges for direct application of traditional backpropagation algorithms. To address this, researchers have utilized surrogate gradients (<xref ref-type="bibr" rid="ref29">Neftci et al., 2019</xref>) to train SNNs using backpropagation methods. During forward propagation, a non-differentiable spiking function is employed, while backpropagation utilizes a continuously differentiable surrogate function for gradient computation. The surrogate gradient of membrane potential is defined by <xref ref-type="disp-formula" rid="EQ4">Equation (4)</xref>:</p>
<disp-formula id="EQ4">
<label>(4)</label>
<mml:math id="M4">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mtext>if</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>U</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mtext>otherwise</mml:mtext>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>The surrogate gradient of the threshold is defined by <xref ref-type="disp-formula" rid="EQ5">Equation (5)</xref>:</p>
<disp-formula id="EQ5">
<label>(5)</label>
<mml:math id="M5">
<mml:mi>H</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>U</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>&#x03B3;</mml:mi>
<mml:mo>&#x22C5;</mml:mo>
<mml:mo>max</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:mi>U</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mo>&#x2223;</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</disp-formula>
<p>The gradient of the weights is defined by <xref ref-type="disp-formula" rid="EQ6">Equations (6&#x2013;8)</xref>:</p>
<disp-formula id="EQ6">
<label>(6)</label>
<mml:math id="M6">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>O</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x22C5;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>O</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>U</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x22C5;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>U</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<disp-formula id="EQ7">
<label>(7)</label>
<mml:math id="M7">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>O</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>U</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>U</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>&#x03B3;</mml:mi>
<mml:mo>&#x22C5;</mml:mo>
<mml:mo>max</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:mi>U</mml:mi>
<mml:mo>&#x2223;</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</disp-formula>
<disp-formula id="EQ8">
<label>(8)</label>
<mml:math id="M8">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>U</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mi>X</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msup>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mtext>if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="0.25em"/>
<mml:mspace width="0.25em"/>
<mml:mtext>if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mi>l</mml:mi>
<mml:mo>&#x003E;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>The gradient of the positive threshold is defined by <xref ref-type="disp-formula" rid="EQ9">Equation (9)</xref>:</p>
<disp-formula id="EQ9">
<label>(9)</label>
<mml:math id="M9">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>+</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x22C5;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>+</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x22C5;</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>U</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>+</mml:mo>
</mml:msup>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</disp-formula>
<p>The gradient of the negative threshold is defined by <xref ref-type="disp-formula" rid="EQ10">Equation (10)</xref>:</p>
<disp-formula id="EQ10">
<label>(10)</label>
<mml:math id="M10">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x22C5;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>U</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</disp-formula>
</sec>
<sec id="sec10">
<label>3.2.2</label>
<title>Quantization scheme and reparameterization technique</title>
<sec id="sec11">
<label>3.2.2.1</label>
<title>First layer quantization scheme</title>
<p>In this paper, we employ 2-bit channel-wise quantization for the first layer weights and 8-bit layer-wise quantization for the remaining layers. Compared to full-precision networks, ternary weight quantization reduces the precision to three discrete values, which is defined by <xref ref-type="disp-formula" rid="EQ11">Equation (11)</xref>:</p>
<disp-formula id="EQ11">
<label>(11)</label>
<mml:math id="M12">
<mml:mi>Q</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="left" displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mtext>if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mi>W</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mtext>if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x003C;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#x003C;</mml:mo>
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mtext>if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mi>W</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math id="M13">
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M14">
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> are the quantization thresholds, with <inline-formula>
<mml:math id="M15">
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x003E;</mml:mo>
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>. We adopt a symmetric form of the quantization thresholds, that is, <italic>&#x03B8;</italic><sub>1</sub>&#x202F;=&#x202F;0.5 and <italic>&#x03B8;</italic><sub>2</sub>&#x202F;=&#x202F;&#x2212;0.5. Then the channel-wise quantization function can be simplified in <xref ref-type="disp-formula" rid="EQ12">Equation (12)</xref>:</p>
<disp-formula id="EQ12">
<label>(12)</label>
<mml:math id="M16">
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable columnalign="left" displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mo mathvariant="italic">sign</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mtext>if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mfrac>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>&#x03B8;</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mtext>otherwise</mml:mtext>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>where <italic>l</italic> indicates the index of the network layer, and <italic>c</italic> represents the index of the output channel of the layer. <italic>f</italic> is the channel scaling factor, which is related to the number of channels in the convolutional kernel. For the convolutional kernel weights <inline-formula>
<mml:math id="M17">
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext mathvariant="italic">out</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext mathvariant="italic">in</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M18">
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext mathvariant="italic">out</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> is the number of output channels, <inline-formula>
<mml:math id="M19">
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext mathvariant="italic">in</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> is the number of input channels, <italic>K</italic> is the kernel size. If <italic>N</italic> is the number of layers in the network, then <inline-formula>
<mml:math id="M20">
<mml:mi>l</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo stretchy="true">}</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext mathvariant="italic">out</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="true">}</mml:mo>
</mml:math>
</inline-formula>.</p>
<p>The channel weight matrix can be defined as <inline-formula>
<mml:math id="M21">
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mtext mathvariant="italic">in</mml:mtext>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. To exclude the influence of outliers, we define the first-layer channel scaling factor <italic>f</italic>, as given in <xref ref-type="disp-formula" rid="EQ13">Equations (13&#x2013;15)</xref>:</p>
<disp-formula id="EQ13">
<label>(13)</label>
<mml:math id="M22">
<mml:msubsup>
<mml:mi>q</mml:mi>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mo>inf</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>&#x211D;</mml:mi>
<mml:mo>&#x2223;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0.99</mml:mn>
<mml:mo stretchy="true">}</mml:mo>
</mml:math>
</disp-formula>
<disp-formula id="EQ14">
<label>(14)</label>
<mml:math id="M23">
<mml:msubsup>
<mml:mi>q</mml:mi>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mo>sup</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>&#x211D;</mml:mi>
<mml:mo>&#x2223;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0.01</mml:mn>
<mml:mo stretchy="true">}</mml:mo>
</mml:math>
</disp-formula>
<disp-formula id="EQ15">
<label>(15)</label>
<mml:math id="M24">
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>max</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:msubsup>
<mml:mi>q</mml:mi>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2223;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:msubsup>
<mml:mi>q</mml:mi>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2223;</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
<mml:mspace width="1em"/>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mo stretchy="true">[</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext mathvariant="italic">out</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="true">]</mml:mo>
</mml:math>
</disp-formula>
<p>where inf represents the infimum and sup represents the supremum. <inline-formula>
<mml:math id="M25">
<mml:msubsup>
<mml:mi>q</mml:mi>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is a threshold indicating that 99% of the channel weight values are less than or equal to this value, while <inline-formula>
<mml:math id="M26">
<mml:msubsup>
<mml:mi>q</mml:mi>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> is a threshold indicating that 1% of the channel weight values are greater than or equal to this value. Due to the non-differentiability of the round and clip, we employ the Straight-Through Estimator (STE) for backpropagation to update the weights (<xref ref-type="bibr" rid="ref2">Bengio et al., 2013</xref>), as shown in <xref ref-type="disp-formula" rid="EQ16">Equation 16</xref>.</p>
<disp-formula id="EQ16">
<label>(16)</label>
<mml:math id="M27">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mtext>if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mo>&#x2223;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mfrac>
<mml:mo>&#x2223;</mml:mo>
<mml:mo>&#x003C;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mtext>otherwise</mml:mtext>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
</sec>
<sec id="sec12">
<label>3.2.2.2</label>
<title>The other layers quantization scheme</title>
<p>All subsequent layers employ 8-bit layer-wise weight quantization, with the quantization function defined by <xref ref-type="disp-formula" rid="EQ17">Equations (17)</xref> and <xref ref-type="disp-formula" rid="EQ18">(18)</xref>:</p>
<disp-formula id="EQ17">
<label>(17)</label>
<mml:math id="M28">
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
<mml:mrow>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x22C5;</mml:mo>
<mml:mtext mathvariant="italic">round</mml:mtext>
<mml:mo stretchy="true">(</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x22C5;</mml:mo>
<mml:mtext mathvariant="italic">clip</mml:mtext>
<mml:mo stretchy="true">(</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mo stretchy="true">[</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo stretchy="true">]</mml:mo>
</mml:math>
</disp-formula>
<disp-formula id="EQ18">
<label>(18)</label>
<mml:math id="M29">
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mtext>if</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:mo>&#x003C;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mtext>otherwise</mml:mtext>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>where <italic>W<sub>l</sub></italic> represents the 32-bit floating-point weights and <italic>b</italic> is the integer bit width set to 8, and thus <italic>S</italic> is 127. Our proposed quantization framework is illustrated in <xref ref-type="fig" rid="fig3">Algorithm 1</xref>.</p>
<fig position="float" id="fig3">
<label>ALGORITHM 1</label>
<caption>
<p>Proposed quantization framework.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g003.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Pseudocode description of training a spiking neural network (SNN) with quantized weights. The process includes epochs and layers iterations, conditional operations for convolutional modules, and forward-backward propagation techniques to update weights using specified formulas. Formulas and operations for quantiles, maximum values, and loss computation are detailed.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec13">
<label>3.2.2.3</label>
<title>Reparameterization technique</title>
<p>Channel-wise quantization is introduced at the first layer, increasing the training overhead. Nonetheless, we use a reparameterization technique that eliminates additional computation during the inference phase. The quantized weights pass through the convolutional layer and BN (Batch Normalization) layer, and the results are:</p>
<disp-formula id="EQ19">
<label>(19)</label>
<mml:math id="M30">
<mml:mtable columnalign="left" displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>&#x03B3;</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>&#x22C5;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:msub>
<mml:mi>&#x03C3;</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mfrac>
<mml:mo>&#x22C5;</mml:mo>
<mml:mtext>float</mml:mtext>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>Q</mml:mi>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>+</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x202F;</mml:mo>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>&#x03B3;</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>&#x03BC;</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
<mml:msub>
<mml:mi>&#x03C3;</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mover accent="true">
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mo>&#x22C5;</mml:mo>
<mml:mtext>float</mml:mtext>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>Q</mml:mi>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>b</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>where <italic>&#x03C3;</italic> is the variance, <italic>&#x03BC;</italic> is the mean, and <italic>&#x03B3;</italic> and <italic>&#x03B2;</italic> are two learnable parameters. <italic>X</italic> represents the input image of the first layer, <italic>QW</italic><sub>1<italic>,c</italic></sub> denotes the quantized ternary weights of the first layer. The term &#x201C;float&#x201D; indicates the conversion from fixed-point to floating-point representation. The data width of <italic>X</italic> is 8-bit, while <italic>QW</italic><sub>1<italic>,c</italic></sub> is quantized into 2-bit, with all other parameters being 32-bit floating-point. The convolutional bias is omitted in this implementation and thus excluded from the <xref ref-type="disp-formula" rid="EQ19">Equation 19</xref>.</p>
<p>As shown in <xref ref-type="disp-formula" rid="EQ19">Equation 19</xref>, <inline-formula>
<mml:math id="M31">
<mml:mover accent="true">
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>and <inline-formula>
<mml:math id="M32">
<mml:mover accent="true">
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> correspond to the output channel of the first layer, and the BN calculation can be converted to a single floating-point multiplication and a single floating-point addition. The computational operations of our dual compensation strategy match those of both layer-wise and channel-wise quantization, as summarized in <xref ref-type="table" rid="tab1">Table 1</xref>. However, channel-wise dual compensation requires additional 4&#x202F;&#x00D7;&#x202F;(<italic>C<sub>out</sub></italic>&#x202F;&#x2212;&#x202F;1)&#x202F;&#x00D7;&#x202F;32-bit storage to maintain the floating-point multiply-add results and positive/negative thresholds. In short, the dual compensation boosts performance without extra computation, needing only more storage.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Comparison of floating-point operations and memory overhead between dual-compensation strategy and traditional quantization methods (<italic>C<sub>out</sub></italic> denotes output channels).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th>Methods</th>
<th align="center" valign="top">Floating-point multiplication</th>
<th align="center" valign="top">Floating-point addition</th>
<th align="center" valign="top">Floating-point storage (32bit)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Layer-wise quantization</td>
<td align="center" valign="top"><italic>C<sub>out</sub></italic></td>
<td align="center" valign="top"><italic>C<sub>out</sub></italic></td>
<td align="center" valign="top">4</td>
</tr>
<tr>
<td align="left" valign="top">Channel-wise quantization</td>
<td align="center" valign="top"><italic>C<sub>out</sub></italic></td>
<td align="center" valign="top"><italic>C<sub>out</sub></italic></td>
<td align="center" valign="top">2<italic>C<sub>out</sub></italic> +&#x202F;2</td>
</tr>
<tr>
<td align="left" valign="top">Proposed</td>
<td align="center" valign="top"><italic>C<sub>out</sub></italic></td>
<td align="center" valign="top"><italic>C<sub>out</sub></italic></td>
<td align="center" valign="top">4<italic>C<sub>out</sub></italic></td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="sec14">
<label>3.3</label>
<title>Design of the unified PE based on FPGA</title>
<sec id="sec15">
<label>3.3.1</label>
<title>Parallelism of unified PEs</title>
<p>Convolutional layers serve as fundamental feature extraction modules in modern deep learning models. A standard convolutional layer computes an output feature map Y (<italic>C<sub>out</sub></italic>&#x202F;&#x00D7;&#x202F;<italic>H&#x2032;</italic>&#x202F;&#x00D7;&#x202F;<italic>W&#x2032;</italic>) by convolving an input feature map X (<italic>C<sub>in</sub></italic>&#x202F;&#x00D7;&#x202F;<italic>H</italic>&#x202F;&#x00D7;&#x202F;<italic>W</italic>) with a filters W (<italic>C<sub>out</sub></italic>&#x202F;&#x00D7;&#x202F;<italic>C<sub>in</sub></italic>&#x202F;&#x00D7;&#x202F;<italic>K</italic>&#x202F;&#x00D7;&#x202F;<italic>K</italic>), formally expressed as given in <xref ref-type="disp-formula" rid="EQ20">Equation (20)</xref>:</p>
<disp-formula id="EQ20">
<label>(20)</label>
<mml:math id="M33">
<mml:mi>Y</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>_</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:munderover>
<mml:mo movablelimits="false">&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mtext mathvariant="italic">in</mml:mtext>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:munderover>
<mml:mo movablelimits="false">&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:munderover>
<mml:mo movablelimits="false">&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:mi>W</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</disp-formula>
<p>where <italic>i</italic> and <italic>j</italic> index the output and input channels respectively, (<italic>m</italic>,<italic>n</italic>) represent the spatial coordinates in the output feature map, <italic>K</italic> is the kernel size, and <italic>b</italic>_<italic>i</italic> represents the bias term. The above operation can be broken down into a cyclic structure as shown in the <xref ref-type="fig" rid="fig4">Figure 3</xref>. The diagram shows the calculation process of the standard convolutional layer and its parallelism. One parallelism represents the sliding the 2D weight kernel across the 2D input feature map. The key parameters and parallelism dimensions are defined as follows:</p>
<fig position="float" id="fig4">
<label>Figure 3</label>
<caption>
<p>Parallel computing architecture and pseudocode implementation of multi-channel convolution.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g004.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Code snippet illustrates nested loops used for convolution operations in neural networks, aligning with a graphical representation. The left shows loops iterating through output channels, input channels, data size, and kernel size. The right features 3D cubes representing input features and weights, indicating parallelism in operations.</alt-text>
</graphic>
</fig>
<p>Input Channels (<italic>C<sub>in</sub></italic>): This parameter defines the input feature map depth, representing the number of stacked 2D feature maps. As shown in the <xref ref-type="fig" rid="fig4">Figure 3</xref>, the &#x201C;Input or Feature&#x201D; cube&#x2019;s depth dimension corresponds to <italic>C<sub>in</sub></italic>; in the pseudocode, <italic>C<sub>in</sub></italic> determines the layer 2 loop boundary, requiring full traversal of all input channels during each output channel computation.</p>
<p>Output Channels (<italic>C<sub>out</sub></italic>): This parameter represents the number of filters in the convolutional layer and determines the depth dimension of the output feature map.</p>
<p>Input Channel Parallelism (<italic>P<sub>in</sub></italic>): This parameter defines the parallel processing capacity across input channels when computing a single output feature map. When <italic>P<sub>in</sub></italic>&#x202F;=&#x202F;<italic>C<sub>in</sub></italic>, as shown in <xref ref-type="fig" rid="fig4">Figure 3</xref>, the convolutional filter simultaneously processes all <italic>C<sub>in</sub></italic> channels and completes accumulation in one computational step. From a hardware implementation perspective, this requires sufficient compute units to process data from all input channels in parallel, thereby speeding up the computation of a single output channel. In our work, the value of the <italic>P<sub>in</sub></italic> is 512.</p>
<p>Output Channel Parallelism (<italic>P<sub>out</sub></italic>): As another parallel computing dimension, it refers to the number of output channels that can be computed at the same time. Since each output channel&#x2019;s computation is independent (e.g., the <italic>i</italic>-th output channel does not depend on the results of the <italic>i</italic>&#x202F;+&#x202F;1 channel), distinct filters can be applied in parallel. As shown in the <xref ref-type="fig" rid="fig4">Figure 3</xref>, when a <italic>P<sub>out</sub></italic> filter is applied to the input feature map in parallel, the Pout output channels can be computed simultaneously. In our implementation, <italic>P<sub>out</sub></italic> is set to 1.</p>
<p>Since the hidden layer&#x2019;s feature map uses 2-bit ternary values {&#x2212;1, 0, 1}, convolution multiplications simplify to additions. The first-layer weights are reduced to 2 bits via ternary quantization. This approach retains the input image at 8-bit precision but enables computing unit reuse through configurable operations.</p>
</sec>
<sec id="sec16">
<label>3.3.2</label>
<title>Design of the unified PE architecture based on FPGA</title>
<p>Based on the above analysis, we implement the unified computing architecture on the FPGA, as shown in <xref ref-type="fig" rid="fig5">Figure 4</xref>. Our SNN acceleration system comprises three key components: data storage, control path and computing core.</p>
<fig position="float" id="fig5">
<label>Figure 4</label>
<caption>
<p>FPGA-based hardware accelerator architecture and overall system for SNNs.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g005.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Diagram of a spiking neural network architecture. Data flows through various components such as DDR, internal memory, and soft core controller via colored arrows representing data, features, and weights input/output. The processing section includes padding, fetch data, pooling, LIF, and BN modules connected to processing elements (PE). The architecture also highlights parallelism of feature and weight processing, showing pathways configured through status registers and command registers. PCIe connects at the input stage, emphasizing integration with external image inputs.</alt-text>
</graphic>
</fig>
<sec id="sec17">
<label>3.3.2.1</label>
<title>Data storage</title>
<p><bold>External Interface:</bold> The system interfaces with the host computer via a Peripheral Component Interconnect Express (PCIe) bus, which serves as the primary channel for raw image data transmission. Double Data Rate SDRAM (DDR) provides high-capacity storage for network weights and intermediate feature maps. The yellow arrows (data input) and red arrows (Data Output) in the diagram indicate the data flow between the DDR and external interfaces.</p>
<p><bold>Internal Memory:</bold> The architecture incorporates on-chip BRAM serving as a data cache. This memory temporarily stores layer-specific weights and feature maps (blue feature input arrows) fetched from DDR, while buffering output feature maps (feature output arrows) processed by PEs.</p>
<p><bold>Storage Control:</bold> This is a scheduling module that manages the transfer of data between the DDR, BRAM, and SNN processing cores. Through Direct Memory Access (DMA) operations, it ensures timely data delivery to the PE array according to execution requirements.</p>
</sec>
<sec id="sec18">
<label>3.3.2.2</label>
<title>Control path</title>
<p><bold>Soft core controller:</bold> The system employs a MicroBlaze processor to orchestrate the inference pipeline, managing network weight loading, PE configuration timing, and coordination with both the storage controller and instruction queue for flexible hardware control.</p>
<p><bold>Instruction queue and decoding:</bold> The controller dispatches predefined instructions to the instruction queue. A dedicated decoding module sequentially decodes these instructions into precise control signals for both the SNN processing core and storage controller. This architecture ensures programmability while supporting diverse SNN architectures.</p>
</sec>
<sec id="sec21">
<label>3.3.2.3</label>
<title>Computing core</title>
<p>The unified computing architecture comprises three key components: PE array, dual data paths, configuration and state control, as shown in <xref ref-type="fig" rid="fig5">Figure 4</xref>.</p>
<p><bold>PE array:</bold> It implements convolution calculations of inputs and weights. It here supports 512 parallelism.</p>
<p><bold>Dual data paths:</bold> These two data paths support 8-bit image input and 2-bit feature input, respectively. In the respective data pathways, the data is fed into the PE array by the necessary pad filling or directly reading (Fetch Data and Padding).</p>
<p><bold>Configuration and state control:</bold> It consists of command registers (Cmd Reg) and status registers (Status Reg). The Cmd Reg receives configuration instructions from the instruction decoder. The Configure signal determines which data path (8-bit BRAM path or 2-bit BRAM path) to enable based on whether the first or subsequent layer is currently being calculated, and directs the corresponding data to the PE array. The Status Reg collects status signals (Status 1/2) from the data path, such as data readiness, computation completion. These statuses are fed back to the soft core controller. This configuration logic is summarized in <xref ref-type="table" rid="tab2">Table 2</xref> and the overall workflow utilizing these configurations is detailed in <xref ref-type="table" rid="tab3">Table 3</xref>.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Configuration logic for dual data path selection.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Computation stage</th>
<th align="left" valign="top">Configure signal</th>
<th align="left" valign="top">Enabled data path</th>
<th align="center" valign="top">Status signal</th>
<th align="left" valign="top">Input type</th>
<th align="left" valign="top">Weight bitwidth</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">First layer</td>
<td align="left" valign="middle">Select Path 1</td>
<td align="left" valign="middle">8-bit BRAM Path</td>
<td align="center" valign="middle">1</td>
<td align="left" valign="middle">8-bit Image</td>
<td align="left" valign="middle">2-bit</td>
</tr>
<tr>
<td align="left" valign="middle">Subsequent layers</td>
<td align="left" valign="middle">Select Path 2</td>
<td align="left" valign="middle">2-bit BRAM Path</td>
<td align="center" valign="middle">2</td>
<td align="left" valign="middle">2-bit Feature</td>
<td align="left" valign="middle">8-bit</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Workflow and state transition mechanism of unified PE.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Step</th>
<th align="left" valign="top">Stage 1: first-layer computation</th>
<th align="left" valign="top">Stage 2: subsequent-layer computation</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">1. Configuration</td>
<td align="left" valign="top">Trigger condition: To process the first layer of the network.<break/>Operation: Cmd Reg issues instructions to switch to Path 1.</td>
<td align="left" valign="top">Trigger condition: To process the second and subsequent layers of the network.<break/>Operation: Cmd Reg issues instructions to switch to Path 2.</td>
</tr>
<tr>
<td align="left" valign="top">2. Data Flow</td>
<td align="left" valign="top">Data source: Original image.<break/>Storage: 8-bit image data loaded into 8-bit BRAM.<break/>Transmission: Data passes through the Padding module and enters PE.</td>
<td align="left" valign="top">Data source: Spike feature maps.<break/>Storage: 2-bit feature data loaded into 2-bit BRAM.<break/>Transmission: Data passes through the Padding module and enters PE.</td>
</tr>
<tr>
<td align="left" valign="top">3. Weight Flow</td>
<td align="left" valign="top">Weight source: First-layer weights (<italic>W</italic><sub>1</sub>).<break/>Transmission: 2-bit weights pass through the Fetch Data module and enters PE.</td>
<td align="left" valign="top">Weight source: Subsequent-layer weights (<italic>W</italic><sub>2</sub>).<break/>Transmission: 8-bit weights pass through the Fetch Data module and enters PE.</td>
</tr>
<tr>
<td align="left" valign="top">4. Core Computation</td>
<td align="left" valign="top">The PE array performs convolution: 8-bit images &#x00D7; 2-bit weights.</td>
<td align="left" valign="top">The PE array performs convolution: 2-bit features &#x00D7; 8-bit weights.</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>After the PE array completes the convolution, the results undergo post-processing including BN and LIF modules to enable neuronal dynamics of the SNN and activate spikes.</p>
<p><bold>BN module:</bold> The convolutional outputs are normalized before entering the neuron model. In hardware implementations, the parameters of the BN, <italic>&#x03B3;</italic> and <italic>&#x03B2;</italic>, are typically fused with convolutional weights. Reparameterization factor <inline-formula>
<mml:math id="M34">
<mml:mover accent="true">
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>and <inline-formula>
<mml:math id="M35">
<mml:mover accent="true">
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> simplifying the BN operation to one multiplication and one addition.</p>
<p><bold>LIF module:</bold> This module receives the value after BN and updates the internal membrane potential according to LIF neural dynamics. The updated membrane potential is compared to a configurable threshold: if the membrane potential exceeds the positive threshold, the neuron emits a spike (+1); if the membrane potential exceeds the negative threshold, the neuron will emits a spike (&#x2212;1); Otherwise, the spike is not activated (0). In this work the time step is set to 1, the membrane potential is not further updated.</p>
<p><bold>Pooling module:</bold> The maximum pooling operation is performed on the spike feature map, reducing spatial dimensions, expanding the receptive field, and improving the model&#x2019;s translation invariance.</p>
<p><bold>Final output:</bold> The results of pooling (the new spike feature map &#x2208; {&#x2212;1, 0, 1}) will be written back to the internal memory as input to the next layer, thus completing a full computation cycle.</p>
<p>In summary, through the innovative configurable dual data path design, the unified PE array can support computing of the first layer (8-bit input) and the subsequent layer (2-bit input) on the same computing arrays. This innovation ensures full reuse of computational resources across all network layers, significantly enhancing hardware utilization.</p>
</sec>
</sec>
</sec>
</sec>
<sec id="sec29">
<label>4</label>
<title>Experiments and discussion</title>
<sec id="sec30">
<label>4.1</label>
<title>Datasets and evaluation metrics</title>
<sec id="sec31">
<label>4.1.1</label>
<title>Dataset</title>
<p>In this study, we utilize CIFAR10 andCIFAR100 datasets. The CIFAR10 dataset (<xref ref-type="bibr" rid="ref20">Krizhevsky and Hinton, 2009</xref>) comprises 60,000 color images measuring 32&#x202F;&#x00D7;&#x202F;32 pixels, distributed across 10 distinct classes. It consists of 50,000 samples for training and an additional 10,000 samples for validation. CIFAR-100 is more challenging. It has 100 classes containing 600 images each.</p>
</sec>
<sec id="sec32">
<label>4.1.2</label>
<title>Evaluation metrics</title>
<p>For performance evaluation, we utilize overall accuracy to assess classification performance. Furthermore, the signal-to-noise ratio (SNR) is used to evaluate the robustness of the quantized model.</p>
</sec>
</sec>
<sec id="sec33">
<label>4.2</label>
<title>Implementation details</title>
<sec id="sec34">
<label>4.2.1</label>
<title>Data preprocessing and networks</title>
<p>For CIFAR10 and CIFAR100 dataset, during training, we apply standard data augmentation techniques, which include adding a 4-pixel padding on each side, performing random 32&#x202F;&#x00D7;&#x202F;32 cropping, and applying random horizontal flipping. However, during validation, the original images are utilized without these techniques. All the images are normalized to achieve a zero mean and unit variance. We employ VGG16, VGG11 (<xref ref-type="bibr" rid="ref34">Simonyan and Zisserman, 2015</xref>) and ResNet19 (<xref ref-type="bibr" rid="ref15">He et al., 2016</xref>) for validation on CIFAR10/100datasets.</p>
</sec>
<sec id="sec35">
<label>4.2.2</label>
<title>Hyperparameters setting</title>
<p>During training, we employ a cross-entropy loss function with stochastic gradient descent optimization which incorporates weight decay (0.0005) and momentum (0.9) parameters. The full precision SNNs are trained for 300 epochs on the all datasets. The learning rate is 0.1 for VGG architectures and 0.01 for ResNet network, with a batch size of 256. We adopt a cosine learning rate decay schedule during training. During quantization stage, the AdamW optimizer is used with a weight decay of 0.01. The leaky factor &#x1D70F; is fixed at 1 and the firing threshold &#x1D703; for ternary spike neurons is initialized at 0.5. The time step is set to 1 for all experiments. We utilize Python 3.10 and PyTorch 1.12 software and two NVIDIA A6000 Graphical Processing Units (GPUs). The operating system is Ubuntu 18.04.</p>
</sec>
<sec id="sec36">
<label>4.2.3</label>
<title>Implementation details of hardware</title>
<p>For the FPGA implementation, we use Verilog and Vivado 2020.2 to design the architecture. The power consumption data comes from the power report provided by the software. The unified PE is deployed on the Xilinx Virtex-7 XC7V690T FPGA operating at 100&#x202F;MHz clock frequency. We adopt the row-stationary strategy used in works like Eyeriss (<xref ref-type="bibr" rid="ref5">Chen et al., 2016</xref>), where kernel-sized rows of the feature map are stored in on-chip cache at the dataflow level. For example, a 3&#x202F;&#x00D7;&#x202F;3 convolution requires caching 3 rows of input data. For each new row processed, the cache updates to present a continuous convolution dataflow to the PEs. This maximizes data reuse and computational efficiency.</p>
</sec>
</sec>
<sec id="sec37">
<label>4.3</label>
<title>Algorithm performance evaluation</title>
<sec id="sec38">
<label>4.3.1</label>
<title>Performance comparison with advanced methods</title>
<p>To evaluate the effectiveness of our method, we conduct comparative experiments with existing quantized SNN approaches on the CIFAR-10 dataset. The results are summarized in <xref ref-type="table" rid="tab4">Table 4</xref>, organized into four cases based on quantization bit-widths for the first layer weights (<italic>W</italic><sub>1</sub>) and subsequent layers weights (<italic>W</italic><sub>2</sub>): 32/32, 8/8, 2/8, and 2/2. The notation &#x201C;<italic>a</italic>/<italic>b</italic>&#x201D; indicates <italic>a</italic>-bit quantization for <italic>W</italic><sub>1</sub> and <italic>b</italic>-bit quantization for <italic>W</italic><sub>2</sub>.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Comparison of the performance of VGG16 and ResNet19 with other SOTA methods on CIFAR10.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Dataset</th>
<th align="left" valign="top">Method</th>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">Precision (<italic>W</italic><sub>1</sub>/<italic>W</italic><sub>2</sub>)</th>
<th align="center" valign="top">Time step</th>
<th align="center" valign="top">Accuracy (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="11">CIFAR10</td>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref41">Yin et al. (2024)</xref>
</td>
<td align="left" valign="top">Vgg16</td>
<td align="center" valign="top">8/8</td>
<td align="center" valign="top">8</td>
<td align="center" valign="top">90.72</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref41">Yin et al. (2024)</xref>
</td>
<td align="left" valign="top">Vgg16</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">8</td>
<td align="center" valign="top">91.15</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref41">Yin et al. (2024)</xref>
</td>
<td align="left" valign="top">ResNet19</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">8</td>
<td align="center" valign="top">91.29</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref41">Yin et al. (2024)</xref>
</td>
<td align="left" valign="top">ResNet19</td>
<td align="center" valign="top">8/8</td>
<td align="center" valign="top">8</td>
<td align="center" valign="top">91.36</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref44">Zhou et al. (2021)</xref>
</td>
<td align="left" valign="top">VGG16</td>
<td align="center" valign="top">2/2</td>
<td align="center" valign="top">-</td>
<td align="center" valign="top">90.93</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref43">Yoo and Jeong (2023)</xref>
</td>
<td align="left" valign="top">VGG16</td>
<td align="center" valign="top">2/2</td>
<td align="center" valign="top">32</td>
<td align="center" valign="top"><bold>91.66</bold></td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref40">Xu et al. (2023)</xref>
</td>
<td align="left" valign="top">VGG16</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">91.05</td>
</tr>
<tr>
<td align="left" valign="top">This work</td>
<td align="left" valign="top">ResNet19</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">91.95</td>
</tr>
<tr>
<td align="left" valign="top">This work</td>
<td align="left" valign="top">ResNet19</td>
<td align="center" valign="top">2/8</td>
<td align="center" valign="top"><bold>1</bold></td>
<td align="center" valign="top"><bold>91.79</bold></td>
</tr>
<tr>
<td align="left" valign="top">This work</td>
<td align="left" valign="top">VGG16</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">91.93</td>
</tr>
<tr>
<td align="left" valign="top">This work</td>
<td align="left" valign="top">VGG16</td>
<td align="center" valign="top">2/8</td>
<td align="center" valign="top"><bold>1</bold></td>
<td align="center" valign="top">91.55</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate superior performance.</p>
</table-wrap-foot>
</table-wrap>
<p>The proposed quantization method achieves low latency and high performance in SNNs. As shown in <xref ref-type="table" rid="tab4">Table 4</xref>, method (<xref ref-type="bibr" rid="ref43">Yoo and Jeong, 2023</xref>) require 32 time steps to reach 91.66% accuracy, whereas our approach attains 91.55% accuracy in single time step, reducing latency by 32&#x202F;&#x00D7;&#x202F;. By optimizing the quantization strategy, our method minimizes quantization loss while maintaining competitive network performance. Experiments on VGG16 and ResNet19 architectures demonstrate accuracies of 91.55% and 91.79%, respectively, outperforming prior results reported by <xref ref-type="bibr" rid="ref42">Yin et al. (2024)</xref> at 90.72% and 91.36%.</p>
<p>To systematically evaluate the effectiveness of our method, we conduct comparative experiments with other SOTA approaches on the CIFAR-100 dataset, as summarized in <xref ref-type="table" rid="tab5">Table 5</xref>.</p>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption>
<p>Comparison of the performance of VGG11 and ResNet19 with other SOTA methods on CIFAR100.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Dataset</th>
<th align="left" valign="top">Method</th>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">Precision (<italic>W</italic><sub>1</sub>/<italic>W</italic><sub>2</sub>)</th>
<th align="center" valign="top">Time step</th>
<th align="center" valign="top">Accuracy (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="11">CIFAR100</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref45">Zou et al. (2024)</xref>
</td>
<td align="left" valign="top">VGG11</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">&#x2013;</td>
<td align="center" valign="top">67.40</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref45">Zou et al. (2024)</xref>
</td>
<td align="left" valign="top">VGG11</td>
<td align="center" valign="top">2/2</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">54.27</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref38">Wang et al. (2025)</xref>
</td>
<td align="left" valign="top">VGG16</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">-</td>
<td align="center" valign="top">77.22</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref38">Wang et al. (2025)</xref>
</td>
<td align="left" valign="top">VGG16</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">2</td>
<td align="center" valign="top">64.89</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref11">Gao et al. (2023)</xref>
</td>
<td align="left" valign="top">VGG16</td>
<td align="center" valign="top">8/8</td>
<td align="center" valign="top">8</td>
<td align="center" valign="top">66.32</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref13">Hasssan et al. (2024)</xref>
</td>
<td align="left" valign="top">ResNet19</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">2</td>
<td align="center" valign="top">72.78</td>
</tr>
<tr>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref13">Hasssan et al. (2024)</xref>
</td>
<td align="left" valign="top">ResNet19</td>
<td align="center" valign="top">4/4</td>
<td align="center" valign="top">2</td>
<td align="center" valign="top">71.87</td>
</tr>
<tr>
<td align="left" valign="top">This work</td>
<td align="left" valign="top">ResNet19</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">72.88</td>
</tr>
<tr>
<td align="left" valign="top">This work</td>
<td align="left" valign="top">ResNet19</td>
<td align="center" valign="top">2/8</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top"><bold>72.23</bold></td>
</tr>
<tr>
<td align="left" valign="top">This work</td>
<td align="left" valign="top">VGG11</td>
<td align="center" valign="top">32/32</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">67.84</td>
</tr>
<tr>
<td align="left" valign="top">This work</td>
<td align="left" valign="top">VGG11</td>
<td align="center" valign="top">2/8</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top"><bold>67.38</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Our method demonstrates clear advantages in inference efficiency, which is especially important in SNNs where the number of time steps (T) directly affects system latency and computational overhead. Experimental results show that the proposed method achieves single-time step inference across multiple architectures, such as ResNet19 and VGG11. In contrast, methods from <xref ref-type="bibr" rid="ref45">Zou et al. (2024)</xref> (T&#x202F;=&#x202F;2), <xref ref-type="bibr" rid="ref13">Hasssan et al. (2024)</xref> (T&#x202F;=&#x202F;2) and <xref ref-type="bibr" rid="ref11">Gao et al. (2023)</xref> (T&#x202F;=&#x202F;8) all require multiple time steps to complete inference. This single-time-step capability makes our approach particularly suitable for latency-sensitive edge computing applications, such as autonomous driving and industrial inspection, where rapid inference is crucial.</p>
<p>In addition, our method maintains high model performance within a single time step. For instance, when quantizing ResNet19 from 32-bit full precision to 2/8-bit, the accuracy only decreases slightly from 72.88 to 72.23%, with a loss of 0.65%. Similarly, On VGG11 declines marginally from 67.84 to 67.38%, with a loss of 0.46%. Compared with the full-precision VGG16 (64.89%) reported by <xref ref-type="bibr" rid="ref38">Wang et al. (2025)</xref>, our approach achieves an improvement of 2.49%. Additionally, compared with the 2-bit quantified VGG11 (54.27%) reported by <xref ref-type="bibr" rid="ref45">Zou et al. (2024)</xref>, our method&#x2019;s performance increases by 13.11%. While the 4-bit quantization scheme (<xref ref-type="bibr" rid="ref13">Hasssan et al., 2024</xref>) achieves 71.87% accuracy in two time steps, our method achieves a comparable performance in one time step. These results demonstrate that our quantization strategy enables high-performance model compression without increasing the number of time steps, providing efficient algorithmic support for hardware-accelerated unified computing architectures.</p>
</sec>
<sec id="sec39">
<label>4.3.2</label>
<title>Hardware-oriented performance trade-off analysis: SNN vs. ANN</title>
<p>To thoroughly assess the proposed method&#x2019;s effectiveness, this section compares our T8HWQ SNN with the W2A8 ANN, a ternary weight network (TWN) (<xref ref-type="bibr" rid="ref24">Liu et al., 2023</xref>) with 8-bit activation quantization. The comparison focuses on accuracy and storage overhead, as shown in <xref ref-type="table" rid="tab6">Table 6</xref>.</p>
<table-wrap position="float" id="tab6">
<label>Table 6</label>
<caption>
<p>Comparison of the proposed SNN and ANN on performance and storage overhead.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" rowspan="2">Dataset</th>
<th align="left" valign="top" rowspan="2">Method</th>
<th align="left" valign="top" rowspan="2">Model</th>
<th align="center" valign="top" rowspan="2"><italic>W</italic><sub>1</sub>/<italic>W</italic><sub>2</sub>/<italic>W<sub>A</sub></italic></th>
<th align="center" valign="top" rowspan="2">Weight (MB)</th>
<th align="center" valign="top" colspan="2">Feature map (KB)</th>
<th align="center" valign="top" rowspan="2">Accuracy (%)</th>
</tr>
<tr>
<th align="center" valign="top">32&#x202F;&#x00D7;&#x202F;32</th>
<th align="center" valign="top">640&#x202F;&#x00D7;&#x202F;640</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="8">CIAFR100</td>
<td align="left" valign="middle">TWN</td>
<td align="left" valign="middle">VGG11(ANN)</td>
<td align="center" valign="middle">32/32/32</td>
<td align="center" valign="middle">35.88</td>
<td align="center" valign="middle">10.66</td>
<td align="center" valign="middle">4262.50</td>
<td align="center" valign="middle"><bold>69.52</bold></td>
</tr>
<tr>
<td align="left" valign="middle">TWN</td>
<td align="left" valign="middle">VGG11(ANN)</td>
<td align="center" valign="middle">2/2/8</td>
<td align="center" valign="middle"><bold>2.24</bold></td>
<td align="center" valign="middle">2.66</td>
<td align="center" valign="middle">1065.62</td>
<td align="center" valign="middle">66.90</td>
</tr>
<tr>
<td align="left" valign="middle">This work</td>
<td align="left" valign="middle">VGG11(SNN)</td>
<td align="center" valign="middle">32/32/2</td>
<td align="center" valign="middle">35.88</td>
<td align="center" valign="middle">0.67</td>
<td align="center" valign="middle">266.41</td>
<td align="center" valign="middle">67.84</td>
</tr>
<tr>
<td align="left" valign="middle">This work</td>
<td align="left" valign="middle">VGG11(SNN)</td>
<td align="center" valign="middle">2/8/2</td>
<td align="center" valign="middle">8.97</td>
<td align="center" valign="middle"><bold>0.67</bold></td>
<td align="center" valign="middle"><bold>266.41</bold></td>
<td align="center" valign="middle"><bold>67.38</bold></td>
</tr>
<tr>
<td align="left" valign="middle">TWN</td>
<td align="left" valign="middle">Resnet19(ANN)</td>
<td align="center" valign="middle">32/32/32</td>
<td align="center" valign="middle">47.64</td>
<td align="center" valign="middle">36.25</td>
<td align="center" valign="middle">14,500</td>
<td align="center" valign="middle"><bold>74.21</bold></td>
</tr>
<tr>
<td align="left" valign="middle">TWN</td>
<td align="left" valign="middle">Resnet19(ANN)</td>
<td align="center" valign="middle">2/2/8</td>
<td align="center" valign="middle"><bold>2.98</bold></td>
<td align="center" valign="middle">9.06</td>
<td align="center" valign="middle">3625.00</td>
<td align="center" valign="middle">71.45</td>
</tr>
<tr>
<td align="left" valign="middle">This work</td>
<td align="left" valign="middle">Resnet19(SNN)</td>
<td align="center" valign="middle">32/32/2</td>
<td align="center" valign="middle">47.64</td>
<td align="center" valign="middle">2.26</td>
<td align="center" valign="middle">906.25</td>
<td align="center" valign="middle">72.88</td>
</tr>
<tr>
<td align="left" valign="middle">This work</td>
<td align="left" valign="middle">Resnet19(SNN)</td>
<td align="center" valign="middle">2/8/2</td>
<td align="center" valign="middle">11.91</td>
<td align="center" valign="middle"><bold>2.26</bold></td>
<td align="center" valign="middle"><bold>906.25</bold></td>
<td align="center" valign="middle"><bold>72.23</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate superior performance.</p>
</table-wrap-foot>
</table-wrap>
<p>While the W2A8 ANN can employ the same unified computing architecture as ours, experimental results demonstrate that our method offers accuracy benefits. On CIFAR-100, with a single time step, our SNN achieves 67.38 and 72.23% accuracy on VGG11 and ResNet19, respectively, surpassing TWN&#x2019;s 66.90 and 71.45%. This indicates that, under ultra-low latency inference constraints, our SNN can still outperform the W2A8 ANN using identical hardware, revealing its higher performance potential.</p>
<p>To improve accuracy, our model balances weight storage against feature map storage. As shown in <xref ref-type="table" rid="tab6">Table 6</xref>, ResNet19&#x2019;s weight storage is 11.91&#x202F;MB with our method, compared to 2.98&#x202F;MB for TWN. Nevertheless, our approach reduces feature map storage and processing overhead. Since our activation values are only 2 bits, feature map storage decreases by approximately 75%, for example, from 9.06&#x202F;KB to 2.26&#x202F;KB in ResNet19. In edge computing chip design, the main performance constraint lies not only in computation but also in data movement. Weight parameters are read once during inference and stored in off-chip DRAM. Conversely, feature maps require frequent read/write operations and must reside in on-chip SRAM to enable low-latency, high-bandwidth data transfer and lower power consumption.</p>
<p>This challenge is particularly prominent in ResNet networks using residual connections: the shallow network&#x2019;s output feature map must be retained on-chip before being added to the deeper feature map after multiple convolutional layers. When on-chip SRAM capacity is insufficient, these feature maps are offloaded to external DRAM and reloaded, incurring substantial latency and power consumption (<xref ref-type="bibr" rid="ref3">Bhati et al., 2016</xref>). Therefore, drastically reducing feature map storage via quantizing activation values to very low bit-widths is a vital strategy to mitigate this challenge and enable efficient hardware acceleration.</p>
<p>This advantage becomes even more pronounced with high-resolution inputs. As shown in the table, the feature map size roughly scales with the square of the input dimension (<italic>N</italic><sup>2</sup>). When the input size increases to 640&#x202F;&#x00D7;&#x202F;640, the ANN&#x2019;s feature map storage rises sharply to 3,625&#x202F;KB, whereas our method requires only 906&#x202F;KB, achieving a fourfold reduction. This scalability demonstrates our approach&#x2019;s strong potential for large-scale data processing, such as high-resolution remote sensing image analysis (<xref ref-type="bibr" rid="ref27">Maggiori et al., 2017</xref>).</p>
<p>In summary, our single-step SNN model outperforms the W2A8 ANN in accuracy with the same computing architecture. Although it reduces weight storage, this trade-off enables improvements in feature map memory, data movement efficiency, and power consumption. This hardware-software co-design offers a promising approach for processing large-scale data on edge devices.</p>
</sec>
<sec id="sec40">
<label>4.3.3</label>
<title>Feature analysis</title>
<p>To systematically explore the information retention ability of the first layer in the quantized model, we employ singular value decomposition (SVD) on the channel characteristic map of this layer. This analysis involves examining the singular values and their cumulative energy distribution, as shown in <xref ref-type="fig" rid="fig6">Figures 5</xref>, <xref ref-type="fig" rid="fig7">6</xref>.</p>
<fig position="float" id="fig6">
<label>Figure 5</label>
<caption>
<p>Feature analysis of VGG16 on CIFAR10 dataset <bold>(a)</bold> singular value distribution comparison between full-precision (w32w32) and quantized (w2w8) models <bold>(b)</bold> cumulative energy ratio and energy difference.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g006.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Graph (a) shows the log of singular value distribution for two SNN models. Blue circles represent the full precision model w32w32, and red crosses represent the quantized model w2w8. Graph (b) illustrates the cumulative energy ratio and energy difference of the SNN models. Blue circles and red crosses show cumulative energy for the full precision and quantized models, respectively. Green squares represent the energy difference.</alt-text>
</graphic>
</fig>
<fig position="float" id="fig7">
<label>Figure 6</label>
<caption>
<p>Feature analysis of ResNet19 on CIFAR10 dataset <bold>(a)</bold> singular value distribution comparison between full-precision (w32w32) and quantized (w2w8) models <bold>(b)</bold> cumulative energy ratio and energy difference.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g007.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Graph (a) shows the singular value distribution of SNN models, comparing full precision and quantized models. Graph (b) illustrates cumulative energy and energy difference, with the latter as a separate line, highlighting small differences between models.</alt-text>
</graphic>
</fig>
<p>The results shown in <xref ref-type="fig" rid="fig6">Figures 5A</xref>, <xref ref-type="fig" rid="fig7">6A</xref> indicate that the proposed quantization scheme successfully retains the key information of the original full-precision model while realizing model compression. Among them, the blue curve representing the full-precision model and the red curve representing the quantization model nearly overlap, demonstrating that their singular value distributions are highly similar. Specifically, as shown in <xref ref-type="fig" rid="fig6">Figure 5</xref> the quantized SNN model closely matches the full-precision model&#x2019;s singular values within the first 25 maximum singular values. Similarly, in <xref ref-type="fig" rid="fig7">Figure 6</xref>, the quantized model exhibits near-identical characteristics for the singular values up to the 95th index. These observations confirm that the quantized model successfully maintains the core information and essential functions of the original full-precision network.</p>
<p>The results displayed in <xref ref-type="fig" rid="fig6">Figures 5B</xref>, <xref ref-type="fig" rid="fig7">6B</xref> demonstrate that the energy distribution of the quantized model closely aligns with that of the full-precision model. Specifically, in <xref ref-type="fig" rid="fig6">Figure 5B</xref>, the top 10 singular values already capture more than 90% of the total energy contribution, with the energy difference within these singular values fluctuating by no more than 0.05. Beyond the 10th singular value, the energy difference diminishes further, remaining below 0.01. Similarly, <xref ref-type="fig" rid="fig7">Figure 6B</xref> shows that the first 20 singular values account for approximately 95% of the energy, and for singular values with indices greater than 20, the energy difference between the quantized and full-precision models is less than 0.005. These findings indicate that the overall energy difference between the full-precision and quantized models is minimal, suggesting that the quantization process effectively preserves the energy distribution of the original model.</p>
<p>Additionally, <xref ref-type="fig" rid="fig6">Figure 5B</xref> illustrates the impact of quantization noise on the model&#x2019;s performance. Specifically, when the singular value index exceeds 25, the red curve representing the quantized model remains above the blue curve of the full-precision model. For these higher-index singular values, which are inherently small, the energy introduced by quantization noise becomes significantly larger than that of the original signal. As a result, the singular values at these locations no longer accurately reflect the fine details of the original model. In particular, at the 26th index, the singular value of the quantized model is markedly larger than that of the full-precision model, indicating that the delicate information contained in the full-precision model has been overwhelmed by noise. This phenomenon suggests that excessive quantization noise at these higher indices can potentially degrade the overall model accuracy by obscuring subtle but important features.</p>
<p>In summary, the quantization approach proposed in this study exhibits both negative and positive effects. On the negative side, quantizing the first layer of the model to 2 bits inherently introduces quantization noise, which can lead to a decline in the overall network performance. Conversely, the analysis based on singular values and energy distributions demonstrates that the proposed method effectively preserves the core functions and essential information of the original model. By accurately maintaining key singular values and the associated energy distributions, the approach ensures that the fundamental capabilities of the network are largely retained. Consequently, this compression strategy successfully reduces model size and complexity while preserving the critical information of the full-precision model, achieving a balance between efficiency and performance.</p>
</sec>
</sec>
<sec id="sec41">
<label>4.4</label>
<title>Robustness evaluation</title>
<p>To illustrate the effects of network architecture and quantization on accuracy under different SNR levels (as outlined in <xref ref-type="fig" rid="fig8">Figure 7</xref>), the corresponding results are presented in <xref ref-type="fig" rid="fig9">Figure 8</xref>. Specifically, at high SNR levels, the accuracy of the w2w8 model closely approaches that of the full-precision model. However, as the SNR decreases, the accuracy of the full-precision model shows a downward trend. For example, at an SNR of 15, the ResNet19 full-precision model achieves approximately 70% accuracy, whereas the w2w8 quantized model maintains a higher accuracy of about 74%. Similarly, for the VGG16 network at the same SNR, the full-precision model&#x2019;s accuracy drops to around 55%, while the w2w8 model retains approximately 62%. These results show that, as the SNR diminishes, the proposed quantization method not only preserves the robustness of the model but exceeds that of the full-precision counterpart, showcasing improved robustness under noisy conditions.</p>
<fig position="float" id="fig8">
<label>Figure 7</label>
<caption>
<p>Impact of noise on image quality under different signal-to-noise ratio (SNR) conditions.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g008.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Three rows of images display a ship, a frog, and an airplane with varying noise levels. From left to right, as the Signal-to-Noise Ratio (SNR) decreases from infinity to 10 decibels, the images progressively become noisier.</alt-text>
</graphic>
</fig>
<fig position="float" id="fig9">
<label>Figure 8</label>
<caption>
<p>Robustness comparison of mixed-precision (w2w8) vs. full-precision (w32w32) on ResNet19 and VGG16 over CIFAR10 under varying SNR conditions.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g009.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Line chart comparing accuracy against SNR for ResNet19 and VGG16 with different configurations. ResNet19_W32 and VGG16_W32 maintain high accuracy up to 35 dB, then decline. ResNet19_W2W8 and VGG16_W2W8 show lower accuracy, with sharp drops from 20 dB.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec42">
<label>4.5</label>
<title>Ablation studies</title>
<p>To investigate the compensatory effect of the proposed neurons on model accuracy, ablation experiments are conducted using the CIFAR-100 dataset with VGG11 and ResNet19 network architectures.</p>
<p>The experimental variables include the proposed neurons and traditional neurons (<xref ref-type="bibr" rid="ref12">Guo et al., 2024</xref>). For the first layer, channel quantization is applied in both the experimental and control groups. In the subsequent layers, layer quantization (<xref ref-type="bibr" rid="ref14">He and Cheng, 2018</xref>), (<xref ref-type="bibr" rid="ref37">Wang et al., 2019</xref>) and traditional neurons (<xref ref-type="bibr" rid="ref12">Guo et al., 2024</xref>) are used. The results are presented in <xref ref-type="table" rid="tab7">Table 7</xref>, while accuracy trends over training epochs are illustrated in <xref ref-type="fig" rid="fig10">Figure 9</xref> for VGG11 and <xref ref-type="fig" rid="fig11">Figure 10</xref> for ResNet19.</p>
<table-wrap position="float" id="tab7">
<label>Table 7</label>
<caption>
<p>Ablation study comparing proposed versus traditional neurons in VGG11 and ResNet19 on CIFAR100 dataset.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Network</th>
<th align="center" valign="top" colspan="2">VGG11</th>
<th align="center" valign="top" colspan="2">ResNet19</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Traditional neuron</td>
<td align="center" valign="middle">&#x2713;</td>
<td align="center" valign="middle">&#x00D7;</td>
<td align="center" valign="middle">&#x2713;</td>
<td align="center" valign="middle">&#x00D7;</td>
</tr>
<tr>
<td align="left" valign="middle">Proposed neuron</td>
<td align="center" valign="middle">&#x00D7;</td>
<td align="center" valign="middle">&#x2713;</td>
<td align="center" valign="middle">&#x00D7;</td>
<td align="center" valign="middle">&#x2713;</td>
</tr>
<tr>
<td align="left" valign="middle">Accuracy</td>
<td align="center" valign="middle">66.68</td>
<td align="center" valign="middle">67.20</td>
<td align="center" valign="middle">71.80</td>
<td align="center" valign="middle">72.20</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig position="float" id="fig10">
<label>Figure 9</label>
<caption>
<p>Training accuracy comparison between proposed and traditional neurons in VGG11 on CIFAR100.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g010.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Line chart comparing accuracy over 50 epochs for two models: Proposed Neuron (red solid line) and Traditional Neuron (blue dashed line). Both models show an overall upward trend from approximately 60% to 68% accuracy, with the Proposed Neuron slightly outperforming the Traditional Neuron in most epochs.</alt-text>
</graphic>
</fig>
<fig position="float" id="fig11">
<label>Figure 10</label>
<caption>
<p>Training accuracy comparison between proposed and traditional neurons in ResNet19 on CIFAR100.</p>
</caption>
<graphic xlink:href="fnins-19-1665778-g011.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Line graph comparing accuracy over 50 epochs for Proposed Neuron and Traditional Neuron. The red line for Proposed Neuron shows consistent improvement, peaking slightly higher than the blue dashed line for Traditional Neuron, indicating similar performance trends with Proposed Neuron achieving slightly better accuracy.</alt-text>
</graphic>
</fig>
<p>The results in <xref ref-type="table" rid="tab7">Table 7</xref> show that the proposed neurons enhance model performance. For the VGG11 model, classification accuracy increases from 66.68% with traditional neurons in the first layer to 67.20% with the proposed neurons, reflecting an improvement of 0.52%. In the ResNet19 model, accuracy in the first layer rises from 71.80% with traditional neurons to 72.20% with the proposed neurons, yielding a 0.40% improvement. These findings confirm that the proposed neurons effectively mitigate the performance loss associated with quantizing the first layer to 2 bits.</p>
<p><xref ref-type="fig" rid="fig10">Figures 9</xref>, <xref ref-type="fig" rid="fig11">10</xref> illustrate that the proposed neurons outperform traditional neurons during training. Initially, the performance of the proposed neurons is approximately 1% lower than that of traditional neurons. However, their performance progressively exceeds that of traditional neurons. In the VGG11 model (<xref ref-type="fig" rid="fig10">Figure 9</xref>), the proposed neurons demonstrate greater stability than traditional neurons starting around the 25th epoch. Similarly, in the ResNet19 model (<xref ref-type="fig" rid="fig11">Figure 10</xref>), the proposed neurons consistently surpass traditional neurons beginning at approximately the 30th epoch.</p>
<p>Overall, the experimental results demonstrate the effectiveness of the neurons introduced in this paper. The proposed neuron successfully compensates for the performance loss associated with 2-bit quantization in the first layer, thereby enhancing the overall performance of the network. This improvement contributes to the development of a high-performance quantization model and offers valuable technical support for hardware unified computing architectures.</p>
</sec>
<sec id="sec43">
<label>4.6</label>
<title>Hardware efficiency evaluation</title>
<p>In this section, we analyze the varying levels of parallelism in PE1. We use the decoupled PE architecture as the benchmark system, and its resource utilization will serve as the baseline for evaluating performance. This comparison will allow us to assess the effectiveness of different degrees of parallelism and their impact on resource utilization.</p>
<p>As shown in <xref ref-type="table" rid="tab8">Table 8</xref>, the unified computing architecture exhibits low resource utilization. By integrating PE1 and PE2, critical logic resources are saved. Specifically, when the output parallelism of PE1 is low (1), the unified PE can save 1.22% of LUTs, 0.49% of flip flops (FFs), and 50% of digital signal processors (DSPs) compared to traditional decoupled PEs. When the output parallelism of PE1 is increased to 16, the unified PE achieves even greater resource savings, with reductions of 20.20% in LUTs, 10.59% in FFs, and 6.90% in DSPs.</p>
<table-wrap position="float" id="tab8">
<label>Table 8</label>
<caption>
<p>Resource comparison between decoupled PE and unified PE for first convolutional layer under different output parallelism.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top"><italic>Pout</italic> (1st layer)</th>
<th align="left" valign="top">Method</th>
<th align="left" valign="top">PE (<italic>P<sub>in</sub></italic> &#x00D7; <italic>P</italic><sub><italic>ou</italic>t</sub>)</th>
<th align="center" valign="top">LUT</th>
<th align="center" valign="top">FF</th>
<th align="center" valign="top">DSP</th>
<th align="center" valign="top">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="2">1</td>
<td align="left" valign="middle">Decoupled PE</td>
<td align="left" valign="middle">PE1(3&#x202F;&#x00D7;&#x202F;1) &#x0026; PE2(512&#x202F;&#x00D7;&#x202F;1)</td>
<td align="center" valign="middle">149,642<break/>(&#x2212;0.00%)</td>
<td align="center" valign="middle">142,212<break/><bold>(</bold>&#x2212;0.00%<bold>)</bold></td>
<td align="center" valign="middle">12<break/><bold>(</bold>&#x2212;0.00%<bold>)</bold></td>
<td align="center" valign="middle" rowspan="2">320</td>
</tr>
<tr>
<td align="left" valign="middle">Unified PE</td>
<td align="left" valign="middle">This work (512&#x202F;&#x00D7;&#x202F;1)</td>
<td align="center" valign="middle"><bold>147,818</bold><break/><bold>(&#x2212;</bold>1.22%<bold>)</bold></td>
<td align="center" valign="middle"><bold>141,516</bold><break/><bold>(&#x2212;</bold>0.49%<bold>)</bold></td>
<td align="center" valign="middle"><bold>6</bold><break/><bold>(&#x2212;50%)</bold></td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="2">16</td>
<td align="left" valign="middle">Decoupled PE</td>
<td align="left" valign="middle">PE1(3&#x202F;&#x00D7;&#x202F;16) &#x0026; PE2(512&#x202F;&#x00D7;&#x202F;1)</td>
<td align="center" valign="middle">199,586<break/><bold>(</bold>&#x2212;0.00%<bold>)</bold></td>
<td align="center" valign="middle">174,634<break/><bold>(</bold>&#x2212;0.00%<bold>)</bold></td>
<td align="center" valign="middle">87<break/><bold>(</bold>&#x2212;0.00%<bold>)</bold></td>
<td align="center" valign="middle" rowspan="2">424</td>
</tr>
<tr>
<td align="left" valign="middle">Unified PE</td>
<td align="left" valign="middle">This work (32&#x202F;&#x00D7;&#x202F;16 &#x0026; 512&#x202F;&#x00D7;&#x202F;1)</td>
<td align="center" valign="middle"><bold>159,263</bold><break/><bold>(&#x2212;20.20%)</bold></td>
<td align="center" valign="middle"><bold>156,137</bold><break/><bold>(&#x2212;10.59%)</bold></td>
<td align="center" valign="middle"><bold>81</bold><break/><bold>(&#x2212;6.90%)</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold values indicate superior performance.</p>
</table-wrap-foot>
</table-wrap>
<p>In addition, the resource-saving benefits of the unified computing architecture are amplified as the degree of parallelism increases. When the parallelism increased from 1 to 16, the savings jumped from 1.22 to 20.20% for LUTs and from 0.49 to 10.59% for FFs. This shows that, compared with the traditional architecture, the unified computing approach can effectively manage the growth of hardware resources in high-parallel real-time tasks.</p>
<p>It&#x2019;s worth noting that the unified computing architecture aligns with the frames per second (FPS) of traditional architectures without compromising processing efficiency. This indicates that the unified computing architecture can achieve substantial resource savings while maintaining the same throughput levels.</p>
<p>In summary, the unified PE represents a more efficient hardware solution than the traditional discrete design. It not only reduces the logic resource utilization of FPGAs but also exhibits a significant scaling effect in high-parallel application scenarios.</p>
<p>In order to fully evaluate the effectiveness of the proposed SNN accelerator design, we conduct a detailed comparison with three advanced similar works on the CIFAR100 dataset. As shown in <xref ref-type="table" rid="tab9">Table 9</xref>, our design exhibits excellent overall performance across several key metrics.</p>
<table-wrap position="float" id="tab9">
<label>Table 9</label>
<caption>
<p>Comparison between FPGA-based SNN accelerator and other SOTA designs.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th>Parameters</th>
<th align="left" valign="top">
<xref ref-type="bibr" rid="ref6">Chen et al. (2024)</xref>
</th>
<th align="left" valign="top">
<xref ref-type="bibr" rid="ref23">Li et al. (2024)</xref>
</th>
<th align="left" valign="top">
<xref ref-type="bibr" rid="ref1">Aung et al. (2023)</xref>
</th>
<th align="left" valign="top">This work</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Platform</td>
<td align="left" valign="top">Virtex-72000&#x202F;T</td>
<td align="left" valign="top">Xczu3eg</td>
<td align="left" valign="top">VCU118</td>
<td align="left" valign="top">Virtex-7690&#x202F;T</td>
</tr>
<tr>
<td align="left" valign="top">Neuron</td>
<td align="left" valign="top">LIF</td>
<td align="left" valign="top">LIF</td>
<td align="left" valign="top">LIF</td>
<td align="left" valign="top">LIF</td>
</tr>
<tr>
<td align="left" valign="top">Dataset</td>
<td align="left" valign="top">CIFAR100</td>
<td align="left" valign="top">CIFAR100</td>
<td align="left" valign="top">CIFAR100</td>
<td align="left" valign="top">CIFAR100</td>
</tr>
<tr>
<td align="left" valign="top">Clock Frequency</td>
<td align="left" valign="top">200</td>
<td align="left" valign="top">600</td>
<td align="left" valign="top">500</td>
<td align="left" valign="top">100</td>
</tr>
<tr>
<td align="left" valign="top">Model</td>
<td align="left" valign="top">VGG11</td>
<td align="left" valign="top">VGG11</td>
<td align="left" valign="top">VGG11</td>
<td align="left" valign="top">VGG11</td>
</tr>
<tr>
<td align="left" valign="top">Weight Bitwidth</td>
<td align="left" valign="top">8bit</td>
<td align="left" valign="top">8bit</td>
<td align="left" valign="top">8bit</td>
<td align="left" valign="top">2bit/8bit</td>
</tr>
<tr>
<td align="left" valign="top">Time Step</td>
<td align="left" valign="top">4</td>
<td align="left" valign="top">4</td>
<td align="left" valign="top">1</td>
<td align="left" valign="top">1</td>
</tr>
<tr>
<td align="left" valign="top">Accuracy</td>
<td align="left" valign="top">66.97%</td>
<td align="left" valign="top">64.3%</td>
<td align="left" valign="top">65.9%</td>
<td align="left" valign="top">67.38%</td>
</tr>
<tr>
<td align="left" valign="top">LUT</td>
<td align="left" valign="top">142,446</td>
<td align="left" valign="top">23&#x202F;K</td>
<td align="left" valign="top">183&#x202F;K</td>
<td align="left" valign="top">147,818</td>
</tr>
<tr>
<td align="left" valign="top">FF</td>
<td align="left" valign="top">124,619</td>
<td align="left" valign="top">&#x2013;</td>
<td align="left" valign="top">&#x2013;</td>
<td align="left" valign="top">141,516</td>
</tr>
<tr>
<td align="left" valign="top">Bram</td>
<td align="left" valign="top">355.0</td>
<td align="left" valign="top">103</td>
<td align="left" valign="top">289</td>
<td align="left" valign="top">326.5</td>
</tr>
<tr>
<td align="left" valign="top">DSP</td>
<td align="left" valign="top">1</td>
<td align="left" valign="top">256</td>
<td align="left" valign="top">2,881</td>
<td align="left" valign="top">6</td>
</tr>
<tr>
<td align="left" valign="top">Latency/Image (ms)</td>
<td align="left" valign="top">19</td>
<td align="left" valign="top">1.75</td>
<td align="left" valign="top">0.082</td>
<td align="left" valign="top">3.12</td>
</tr>
<tr>
<td align="left" valign="top">FPS</td>
<td align="left" valign="top">52</td>
<td align="left" valign="top">571</td>
<td align="left" valign="top">11.6&#x202F;K</td>
<td align="left" valign="top">320</td>
</tr>
<tr>
<td align="left" valign="top">Power Consumption</td>
<td align="left" valign="top">1.562</td>
<td align="left" valign="top">6.2</td>
<td align="left" valign="top">29.8</td>
<td align="left" valign="top">0.982</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Firstly, regarding classification accuracy, the proposed method achieves 67.38%, surpassing all comparison works. This outcome indicates that the T8HWQ scheme and the network architecture can be effectively deployed while maintaining high accuracy. Importantly, this level of performance is achieved at a single time step, without introducing any additional latency. Therefore, our method offers the dual benefits of low latency and high performance.</p>
<p>Secondly, the efficiency of this design is reflected in two key aspects: power consumption and resource utilization. On the Virtex-7690&#x202F;T platform, the power consumption of the proposed accelerator is 0.982&#x202F;W, representing a significant improvement over the 1.562&#x202F;W reported in <xref ref-type="bibr" rid="ref6">Chen et al. (2024)</xref>. This advantage stems not only from the low-latency design achieved within a single time step but also from the implementation of first-layer ternary quantization technology. This technology reduces the first-layer multiplication operation to an equivalent addition operation, drastically decreasing dependence on DSPs. Specifically, our design requires only 6 DSPs, in contrast to 256 and 2,881 DSPs required by the schemes in <xref ref-type="bibr" rid="ref23">Li et al. (2024)</xref> and <xref ref-type="bibr" rid="ref1">Aung et al. (2023)</xref>, respectively. These results demonstrate the applicability of our architecture in resource-constrained edge computing scenarios.</p>
<p>Regarding throughput, while the methods presented in <xref ref-type="bibr" rid="ref23">Li et al. (2024)</xref> and <xref ref-type="bibr" rid="ref1">Aung et al. (2023)</xref> achieve lower latency with higher clock frequencies (600&#x202F;MHz and 500&#x202F;MHz), these performance gains come at the cost of substantial power consumption and DSP resource utilization. In comparison, our design achieves an image processing delay of 3.12&#x202F;ms and a throughput rate of 320 FPS at a clock frequency of only 100&#x202F;MHz. The processing latency of our proposed method is approximately six times lower than that of <xref ref-type="bibr" rid="ref6">Chen et al. (2024)</xref>, despite the latter&#x2019;s implementation being deployed for four time steps, which is four times that of our design. This indicates that our method possesses a highly competitive high-throughput characteristic.</p>
<p>In summary, this study demonstrates that through the software-hardware co-design strategy, the T8HWQ quantization method effectively facilitates the efficient reuse of computing resources. It achieves high accuracy at a single time step while maintaining low levels of power consumption and DSP resource utilization, making it well-suited for resource-constrained, low-latency edge computing scenarios.</p>
</sec>
<sec id="sec44">
<label>4.7</label>
<title>Discussion on scalability for multi-timestep processing</title>
<p>Although the proposed design targets single-timestep scenarios for ultra-low latency processing, it can also scale to multi-timestep applications. Since our architecture does not rely on membrane potential states dependent on specific time steps, the simplest scaling approach involves adopting a temporal parallelism strategy (<xref ref-type="bibr" rid="ref41">Yin et al., 2024</xref>), (<xref ref-type="bibr" rid="ref6">Chen et al., 2024</xref>). In this approach, independent computing resources are allocated for each time step, thereby preserving extremely low latency.</p>
<p>This parallelism strategy highlights a fundamental trade-off in SNN accelerator design: the performance advantage of &#x201C;temporal parallelism&#x201D; versus the resource efficiency of &#x201C;temporal serialism&#x201D; (<xref ref-type="bibr" rid="ref28">Narayanan et al., 2020</xref>). A more flexible and desirable approach involves a hybrid, configurable data flow (<xref ref-type="bibr" rid="ref21">Lee et al., 2022</xref>) that dynamically balances latency and resource utilization based on specific application requirements. For example, for a task with six time steps (T&#x202F;=&#x202F;6), the system can operate in a mode that processes time steps in parallel within a group and executes groups sequentially. This can be realized as three stages, each processing two time steps (T&#x202F;=&#x202F;2) in parallel. Alternatively, it can run in two sequential stages, each processing three time steps (T&#x202F;=&#x202F;3) in parallel.</p>
<p>The T8HWQ unified computing architecture proposed in this paper provides a robust foundation for achieving this goal. It minimizes hardware overhead while maintaining high performance through co-design of software and hardware, which makes it feasible to implement configurable data flows on resource-constrained platforms. In the future, we plan to develop and validate the implementation of this configurable data flow.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec45">
<label>5</label>
<title>Conclusion</title>
<p>This paper addresses the critical issue of resource redundancy in SNN accelerators, a problem stemming from the inherent decoupling of quantization algorithms and FPGA computing units, by proposing a holistic software-hardware co-design methodology. Specifically, we propose a T8HWQ method and a channel-wise dual compensation strategy, which innovatively introduces channel-wise adaptive thresholds to compensate for quantization loss, and adopts a reparameterization method to reduce the quantization performance loss without increasing computation amount. In addition, this proposed method effectively reduces the hardware implementation overhead by supporting a unified computing architecture based on FPGA. Experimental results show that the quantization algorithm and the hardware design not only maintain the high performance of the network quantization, but also realize the resource reuse of computing units in one time step. This algorithm-hardware collaborative optimization scheme provides effective technical support for high-performance and low-latency processing in resource-constrained scenarios. On this basis, we will further explore the algorithm design and hardware architecture development of SNNs in more complex tasks such as object detection in the future (<xref ref-type="bibr" rid="ref6">Chen et al., 2024</xref>).</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec46">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author/s.</p>
</sec>
<sec sec-type="author-contributions" id="sec47">
<title>Author contributions</title>
<p>JL: Software, Validation, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing, Conceptualization, Formal analysis, Methodology. MX: Formal analysis, Validation, Writing &#x2013; review &#x0026; editing, Methodology. HD: Software, Validation, Writing &#x2013; review &#x0026; editing. BL: Software, Validation, Writing &#x2013; review &#x0026; editing. YL: Writing &#x2013; review &#x0026; editing, Investigation. HC: Funding acquisition, Resources, Writing &#x2013; review &#x0026; editing, Supervision. YZ: Writing &#x2013; review &#x0026; editing, Supervision. YX: Writing &#x2013; review &#x0026; editing. LC: Funding acquisition, Writing &#x2013; review &#x0026; editing, Supervision.</p>
</sec>
<sec sec-type="funding-information" id="sec48">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the Foundation under Grant JCKY2021602B037.</p>
</sec>
<sec sec-type="COI-statement" id="sec49">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec50">
<title>Generative AI statement</title>
<p>The authors declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec51">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aung</surname><given-names>M. T. L.</given-names></name> <name><surname>Gerlinghoff</surname><given-names>D.</given-names></name> <name><surname>Qu</surname><given-names>C.</given-names></name> <name><surname>Yang</surname><given-names>L.</given-names></name> <name><surname>Huang</surname><given-names>T.</given-names></name> <name><surname>Goh</surname><given-names>R. S. M.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Deepfire2: a convolutional spiking neural network accelerator on FPGAs</article-title>. <source>IEEE Trans. Comput.</source> <volume>72</volume>, <fpage>2847</fpage>&#x2013;<lpage>2857</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TC.2023.3272284</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bengio</surname><given-names>Y.</given-names></name> <name><surname>L&#x00E9;onard</surname><given-names>N.</given-names></name> <name><surname>Courville</surname><given-names>A.</given-names></name></person-group> (<year>2013</year>). <article-title>Estimating or propagating gradients through stochastic neurons for conditional computation</article-title>. <source>arxiv</source> <volume>2013</volume>:<fpage>3432</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1308.3432</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bhati</surname><given-names>I.</given-names></name> <name><surname>Chang</surname><given-names>M.-T.</given-names></name> <name><surname>Chishti</surname><given-names>Z.</given-names></name> <name><surname>Lu</surname><given-names>S.-L.</given-names></name> <name><surname>Jacob</surname><given-names>B.</given-names></name></person-group> (<year>2016</year>). <article-title>DRAM refresh mechanisms, penalties, and trade-offs</article-title>. <source>IEEE Trans. Comput.</source> <volume>65</volume>, <fpage>108</fpage>&#x2013;<lpage>121</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TC.2015.2417540</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cao</surname><given-names>H.</given-names></name> <name><surname>Zhou</surname><given-names>Z.</given-names></name> <name><surname>Wei</surname><given-names>W.</given-names></name> <name><surname>Belatreche</surname><given-names>A.</given-names></name> <name><surname>Liang</surname><given-names>Y.</given-names></name> <name><surname>Zhang</surname><given-names>D.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Binary event-driven spiking transformer</article-title>. <source>arxiv</source> <volume>2025</volume>:<fpage>5904</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2501.05904</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>Y. H.</given-names></name> <name><surname>Emer</surname><given-names>J.</given-names></name> <name><surname>Sze</surname><given-names>V.</given-names></name></person-group> (<year>2016</year>). <italic>Eyeriss: a spatial architecture for energy-efficient dataflow for convolutional neural networks</italic>. in 2016 ACM/IEEE 43rd annual international symposium on computer architecture (ISCA), (Seoul, South Korea: IEEE), pp. 367&#x2013;379.</citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>Y.</given-names></name> <name><surname>Ye</surname><given-names>W.</given-names></name> <name><surname>Liu</surname><given-names>Y.</given-names></name> <name><surname>Zhou</surname><given-names>H.</given-names></name></person-group> (<year>2024</year>). <article-title>Sibrain: a sparse spatio-temporal parallel neuromorphic architecture for accelerating spiking convolution neural networks with low latency</article-title>. <source>IEEE Trans. Circ. Syst.</source> <volume>71</volume>, <fpage>6482</fpage>&#x2013;<lpage>6494</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TCSI.2024.3393233</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Chowdhury</surname><given-names>S. S.</given-names></name> <name><surname>Garg</surname><given-names>I.</given-names></name> <name><surname>Roy</surname><given-names>K.</given-names></name></person-group> (<year>2021</year>). <italic>Spatio-temporal pruning and quantization for low-latency spiking neural networks</italic>. In 2021 international joint conference on neural networks (IJCNN), pp. 1&#x2013;9.</citation></ref>
<ref id="ref8"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Courbariaux</surname><given-names>M.</given-names></name> <name><surname>Bengio</surname><given-names>Y.</given-names></name> <name><surname>David</surname><given-names>J. P.</given-names></name></person-group> (<year>2015</year>). <italic>BinaryConnect: training deep neural networks with binary weights during propagations</italic>. in Advances in neural information processing systems. Available online at: <ext-link xlink:href="https://proceedings.neurips.cc/paper_files/paper/2015/file/3e15cc11f979ed25912dff5b0669f2cd-Paper.pdf" ext-link-type="uri">https://proceedings.neurips.cc/paper_files/paper/2015/file/3e15cc11f979ed25912dff5b0669f2cd-Paper.pdf</ext-link>.</citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname><given-names>L.</given-names></name> <name><surname>Wu</surname><given-names>Y.</given-names></name> <name><surname>Hu</surname><given-names>Y.</given-names></name> <name><surname>Liang</surname><given-names>L.</given-names></name> <name><surname>Li</surname><given-names>G.</given-names></name> <name><surname>Hu</surname><given-names>X.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Comprehensive SNN compression using ADMM optimization and activity regularization</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>34</volume>, <fpage>2791</fpage>&#x2013;<lpage>2805</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TNNLS.2021.3109064</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eshraghian</surname><given-names>J. K.</given-names></name> <name><surname>Lu</surname><given-names>W. D.</given-names></name></person-group> (<year>2022</year>). <article-title>The fine line between dead neurons and sparsity in binarized spiking neural networks</article-title>. <source>arXiv</source> <volume>2022</volume>:<fpage>11915</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2201.11915</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname><given-names>H.</given-names></name> <name><surname>He</surname><given-names>J.</given-names></name> <name><surname>Wang</surname><given-names>H.</given-names></name> <name><surname>Wang</surname><given-names>T.</given-names></name> <name><surname>Zhong</surname><given-names>Z.</given-names></name> <name><surname>Yu</surname><given-names>J.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>High-accuracy deep ANN-to-SNN conversion using quantization-aware training framework and calcium-gated bipolar leaky integrate and fire neuron</article-title>. <source>Front. Neurosci.</source> <volume>17</volume>:<fpage>1701</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fnins.2023.1141701</pub-id>, PMID: <pub-id pub-id-type="pmid">36968504</pub-id></citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname><given-names>Y.</given-names></name> <name><surname>Chen</surname><given-names>Y.</given-names></name> <name><surname>Liu</surname><given-names>X.</given-names></name> <name><surname>Peng</surname><given-names>W.</given-names></name> <name><surname>Zhang</surname><given-names>Y.</given-names></name> <name><surname>Huang</surname><given-names>X.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Ternary spike: learning ternary spikes for spiking neural networks</article-title>. <source>Proc. AAAI Confe. Artif. Intell.</source> <volume>38</volume>, <fpage>12244</fpage>&#x2013;<lpage>12252</lpage>. doi: <pub-id pub-id-type="doi">10.1609/aaai.v38i11.29114</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hasssan</surname><given-names>A.</given-names></name> <name><surname>Meng</surname><given-names>J.</given-names></name> <name><surname>Anupreetham</surname><given-names>A.</given-names></name> <name><surname>Seo</surname><given-names>J.</given-names></name></person-group> (<year>2024</year>). <article-title>SpQuant-SNN: ultra-low precision membrane potential with sparse activations unlock the potential of on-device spiking neural networks applications</article-title>. <source>Front. Neurosci.</source> <volume>18</volume>:<fpage>144</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fnins.2024.1440000</pub-id>, PMID: <pub-id pub-id-type="pmid">39296710</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="book"><person-group person-group-type="author"><name><surname>He</surname><given-names>X.</given-names></name> <name><surname>Cheng</surname><given-names>J.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>Learning compression from limited unlabeled data</article-title>&#x201D; in <source>Computer vision &#x2013; ECCV 2018</source>. eds. <person-group person-group-type="editor"><name><surname>Ferrari</surname><given-names>V.</given-names></name> <name><surname>Hebert</surname><given-names>M.</given-names></name> <name><surname>Sminchisescu</surname><given-names>C.</given-names></name> <name><surname>Weiss</surname><given-names>Y.</given-names></name></person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>778</fpage>&#x2013;<lpage>795</lpage>.</citation></ref>
<ref id="ref15"><citation citation-type="other"><person-group person-group-type="author"><name><surname>He</surname><given-names>K.</given-names></name> <name><surname>Zhang</surname><given-names>X.</given-names></name> <name><surname>Ren</surname><given-names>S.</given-names></name> <name><surname>Sun</surname><given-names>J.</given-names></name></person-group> (<year>2016</year>). <italic>Deep residual learning for image recognition</italic>. in 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770&#x2013;778.</citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname><given-names>G.</given-names></name> <name><surname>Vinyals</surname><given-names>O.</given-names></name> <name><surname>Dean</surname><given-names>J.</given-names></name></person-group> (<year>2015</year>). <article-title>Distilling the knowledge in a neural network</article-title>. <source>arXiv</source> <volume>2015</volume>:<fpage>2531</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1503.02531</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Jacob</surname><given-names>B.</given-names></name> <name><surname>Kligys</surname><given-names>S.</given-names></name> <name><surname>Chen</surname><given-names>B.</given-names></name> <name><surname>Zhu</surname><given-names>M.</given-names></name> <name><surname>Tang</surname><given-names>M.</given-names></name> <name><surname>Howard</surname><given-names>A.</given-names></name> <etal/></person-group>. (<year>2018</year>). <italic>Quantization and training of neural networks for efficient integer-arithmetic-only inference</italic>. in 2018 IEEE/CVF conference on computer vision and pattern recognition, pp. 2704&#x2013;2713.</citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Karamimanesh</surname><given-names>M.</given-names></name> <name><surname>Abiri</surname><given-names>E.</given-names></name> <name><surname>Shahsavari</surname><given-names>M.</given-names></name> <name><surname>Hassanli</surname><given-names>K.</given-names></name> <name><surname>van Schaik</surname><given-names>A.</given-names></name> <name><surname>Eshraghian</surname><given-names>J.</given-names></name></person-group> (<year>2025</year>). <article-title>Spiking neural networks on FPGA: a survey of methodologies and recent advancements</article-title>. <source>Neural Netw.</source> <volume>186</volume>:<fpage>107256</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neunet.2025.107256</pub-id>, PMID: <pub-id pub-id-type="pmid">39965527</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krishnamoorthi</surname><given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>Quantizing deep convolutional networks for efficient inference: a whitepaper</article-title>. <source>arXiv</source> <volume>2018</volume>:<fpage>8342</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1806.08342</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Krizhevsky</surname><given-names>A.</given-names></name> <name><surname>Hinton</surname><given-names>G.</given-names></name></person-group> (<year>2009</year>). <source>Learning multiple layers of features from tiny images</source>. <publisher-loc>Toronto, Ontario</publisher-loc>: <publisher-name>University of Toronto</publisher-name>.</citation></ref>
<ref id="ref21"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Lee</surname><given-names>J. J.</given-names></name> <name><surname>Zhang</surname><given-names>W.</given-names></name> <name><surname>Li</surname><given-names>P.</given-names></name></person-group> (<year>2022</year>). <italic>Parallel time batching: systolic-Array acceleration of sparse spiking neural computation</italic>. in 2022 IEEE international symposium on high-performance computer architecture (HPCA), pp. 317&#x2013;330.</citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>H.</given-names></name> <name><surname>Kadav</surname><given-names>A.</given-names></name> <name><surname>Durdanovic</surname><given-names>I.</given-names></name> <name><surname>Samet</surname><given-names>H.</given-names></name> <name><surname>Graf</surname><given-names>H. P.</given-names></name></person-group> (<year>2017</year>). <article-title>Pruning filters for efficient ConvNets</article-title>. <source>arXiv</source> <volume>2017</volume>:<fpage>710</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1608.08710</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>J.</given-names></name> <name><surname>Shen</surname><given-names>G.</given-names></name> <name><surname>Zhao</surname><given-names>D.</given-names></name> <name><surname>Zhang</surname><given-names>Q.</given-names></name> <name><surname>Zeng</surname><given-names>Y.</given-names></name></person-group> (<year>2024</year>). <article-title>Firefly v2: advancing hardware support for high-performance spiking neural network with a spatiotemporal FPGA accelerator</article-title>. <source>IEEE Trans. Comput. Aided Des. Integr. Circuits Syst.</source> <volume>43</volume>, <fpage>2647</fpage>&#x2013;<lpage>2660</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TCAD.2024.3380550</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>B.</given-names></name> <name><surname>Li</surname><given-names>F.</given-names></name> <name><surname>Wang</surname><given-names>X.</given-names></name> <name><surname>Zhang</surname><given-names>B.</given-names></name> <name><surname>Yan</surname><given-names>J.</given-names></name></person-group> (<year>2023</year>). <italic>Ternary weight networks</italic>. in ICASSP 2023&#x2013;2023 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp. 1&#x2013;5.</citation></ref>
<ref id="ref25"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Luo</surname><given-names>X.</given-names></name> <name><surname>Yao</surname><given-names>M.</given-names></name> <name><surname>Chou</surname><given-names>Y.</given-names></name> <name><surname>Xu</surname><given-names>B.</given-names></name> <name><surname>Li</surname><given-names>G.</given-names></name></person-group> (<year>2025</year>). &#x201C;<article-title>Integer-valued training and spike-driven inference spiking neural network for high-performance and energy-efficient object detection</article-title>&#x201D; in <source>Computer vision &#x2013; ECCV 2024</source>. eds. <person-group person-group-type="editor"><name><surname>Leonardis</surname><given-names>A.</given-names></name> <name><surname>Ricci</surname><given-names>E.</given-names></name> <name><surname>Roth</surname><given-names>S.</given-names></name> <name><surname>Russakovsky</surname><given-names>O.</given-names></name> <name><surname>Sattler</surname><given-names>T.</given-names></name> <name><surname>Varol</surname><given-names>G.</given-names></name></person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>), <fpage>253</fpage>&#x2013;<lpage>272</lpage>.</citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maass</surname><given-names>W.</given-names></name></person-group> (<year>1997</year>). <article-title>Networks of spiking neurons: the third generation of neural network models</article-title>. <source>Neural Netw.</source> <volume>10</volume>, <fpage>1659</fpage>&#x2013;<lpage>1671</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0893-6080(97)00011-7</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maggiori</surname><given-names>E.</given-names></name> <name><surname>Tarabalka</surname><given-names>Y.</given-names></name> <name><surname>Charpiat</surname><given-names>G.</given-names></name> <name><surname>Alliez</surname><given-names>P.</given-names></name></person-group> (<year>2017</year>). <article-title>Convolutional neural networks for large-scale remote-sensing image classification</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>55</volume>, <fpage>645</fpage>&#x2013;<lpage>657</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TGRS.2016.2612821</pub-id></citation></ref>
<ref id="ref28"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Narayanan</surname><given-names>S.</given-names></name> <name><surname>Taht</surname><given-names>K.</given-names></name> <name><surname>Balasubramonian</surname><given-names>R.</given-names></name> <name><surname>Giacomin</surname><given-names>E.</given-names></name> <name><surname>Gaillardon</surname><given-names>P. E.</given-names></name></person-group> (<year>2020</year>). <italic>SpinalFlow: An Architecture and Dataflow Tailored for Spiking Neural Networks</italic>. in 2020 ACM/IEEE 47th Annual International Symposium on Computer Architecture (ISCA), (Valencia, Spain: IEEE), pp. 349&#x2013;362.</citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Neftci</surname><given-names>E. O.</given-names></name> <name><surname>Mostafa</surname><given-names>H.</given-names></name> <name><surname>Zenke</surname><given-names>F.</given-names></name></person-group> (<year>2019</year>). <article-title>Surrogate gradient learning in spiking neural networks: bringing the power of gradient-based optimization to spiking neural networks</article-title>. <source>IEEE Signal Process. Mag.</source> <volume>36</volume>, <fpage>51</fpage>&#x2013;<lpage>63</lpage>. doi: <pub-id pub-id-type="doi">10.1109/MSP.2019.2931595</pub-id></citation></ref>
<ref id="ref30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Plagwitz</surname><given-names>P.</given-names></name> <name><surname>Hannig</surname><given-names>F.</given-names></name> <name><surname>Teich</surname><given-names>J.</given-names></name> <name><surname>Keszocze</surname><given-names>O.</given-names></name></person-group> (<year>2023</year>). <article-title>To spike or not to spike? A quantitative comparison of SNN and CNN FPGA implementations</article-title>. <source>arXiv</source> <volume>2023</volume>:<fpage>12742</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2306.12742</pub-id></citation></ref>
<ref id="ref31"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Putra</surname><given-names>R. V. W.</given-names></name> <name><surname>Shafique</surname><given-names>M.</given-names></name></person-group> (<year>2021</year>). <italic>Q-SpiNN: a framework for quantizing spiking neural networks</italic>. in 2021 international joint conference on neural networks (IJCNN), pp. 1&#x2013;8.</citation></ref>
<ref id="ref33"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Shymyrbay</surname><given-names>A.</given-names></name> <name><surname>Fouda</surname><given-names>M. E.</given-names></name> <name><surname>Eltawil</surname><given-names>A.</given-names></name></person-group> (<year>2023</year>). <italic>Low precision quantization-aware training in spiking neural networks with differentiable quantization function</italic>. in 2023 international joint conference on neural networks (IJCNN), pp. 1&#x2013;8.</citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simonyan</surname><given-names>K.</given-names></name> <name><surname>Zisserman</surname><given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <source>arXiv</source> <volume>2015</volume>:<fpage>1556</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1409.1556</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Su</surname><given-names>Q.</given-names></name> <name><surname>Chou</surname><given-names>Y.</given-names></name> <name><surname>Hu</surname><given-names>Y.</given-names></name> <name><surname>Li</surname><given-names>J.</given-names></name> <name><surname>Mei</surname><given-names>S.</given-names></name> <name><surname>Zhang</surname><given-names>Z.</given-names></name> <etal/></person-group>. (<year>2023</year>). <italic>Deep directly-trained spiking neural networks for object detection</italic>. in 2023 IEEE/CVF international conference on computer vision (ICCV), pp. 6532&#x2013;6542.</citation></ref>
<ref id="ref36"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Venkatesh</surname><given-names>S.</given-names></name> <name><surname>Marinescu</surname><given-names>R.</given-names></name> <name><surname>Eshraghian</surname><given-names>J. K.</given-names></name></person-group> (<year>2024</year>). <italic>SQUAT: Stateful quantization-aware training in recurrent spiking neural networks</italic>. in 2024 neuro inspired computational elements conference (NICE), pp. 1&#x2013;10.</citation></ref>
<ref id="ref37"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>K.</given-names></name> <name><surname>Liu</surname><given-names>Z.</given-names></name> <name><surname>Lin</surname><given-names>Y.</given-names></name> <name><surname>Lin</surname><given-names>J.</given-names></name> <name><surname>Han</surname><given-names>S.</given-names></name></person-group> (<year>2019</year>). <italic>HAQ: hardware-aware automated quantization with mixed precision</italic>. in 2019 IEEE/CVF conference on computer vision and pattern recognition (CVPR), pp. 8604&#x2013;8612.</citation></ref>
<ref id="ref38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>Z.</given-names></name> <name><surname>Zhang</surname><given-names>Y.</given-names></name> <name><surname>Lian</surname><given-names>S.</given-names></name> <name><surname>Cui</surname><given-names>X.</given-names></name> <name><surname>Yan</surname><given-names>R.</given-names></name> <name><surname>Tang</surname><given-names>H.</given-names></name></person-group> (<year>2025</year>). <article-title>Toward high-accuracy and low-latency spiking neural networks with two-stage optimization</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>36</volume>, <fpage>3189</fpage>&#x2013;<lpage>3203</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TNNLS.2023.3337176</pub-id></citation></ref>
<ref id="ref39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname><given-names>L.</given-names></name> <name><surname>Ma</surname><given-names>Z.</given-names></name> <name><surname>Yang</surname><given-names>C.</given-names></name> <name><surname>Yao</surname><given-names>Q.</given-names></name></person-group> (<year>2024</year>). <article-title>Advances in the neural network quantization: a comprehensive review</article-title>. <source>Appl. Sci.</source> <volume>14</volume>:<fpage>7445</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app14177445</pub-id></citation></ref>
<ref id="ref40"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Xu</surname><given-names>Q.</given-names></name> <name><surname>Li</surname><given-names>Y.</given-names></name> <name><surname>Shen</surname><given-names>J.</given-names></name> <name><surname>Liu</surname><given-names>J. K.</given-names></name> <name><surname>Tang</surname><given-names>H.</given-names></name> <name><surname>Pan</surname><given-names>G.</given-names></name></person-group> (<year>2023</year>). <italic>Constructing deep spiking neural networks from artificial neural networks with knowledge distillation</italic>. in Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition, pp. 7886&#x2013;7895.</citation></ref>
<ref id="ref41"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Yin</surname><given-names>R.</given-names></name> <name><surname>Kim</surname><given-names>Y.</given-names></name> <name><surname>Wu</surname><given-names>D.</given-names></name> <name><surname>Panda</surname><given-names>P.</given-names></name></person-group> (<year>2024</year>). <italic>LoAS: fully temporal-parallel dataflow for dual-sparse spiking neural networks</italic>. in 2024 57th IEEE/ACM international symposium on microarchitecture (MICRO), pp. 1107&#x2013;1121.</citation></ref>
<ref id="ref42"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Yin</surname><given-names>R.</given-names></name> <name><surname>Li</surname><given-names>Y.</given-names></name> <name><surname>Moitra</surname><given-names>A.</given-names></name> <name><surname>Panda</surname><given-names>P.</given-names></name></person-group> (<year>2024</year>). <italic>MINT: Multiplier-less INTeger quantization for energy efficient spiking neural networks</italic>. in 2024 29th Asia and South Pacific design automation conference (ASP-DAC), (Incheon, Korea, republic of: IEEE), pp. 830&#x2013;835.</citation></ref>
<ref id="ref43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yoo</surname><given-names>D.</given-names></name> <name><surname>Jeong</surname><given-names>D. S.</given-names></name></person-group> (<year>2023</year>). <article-title>CBP-QSNN: spiking neural networks quantized using constrained backpropagation</article-title>. <source>IEEE J. Emerg. Sel. Topics Circuits Syst.</source> <volume>13</volume>, <fpage>1137</fpage>&#x2013;<lpage>1146</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JETCAS.2023.3328911</pub-id></citation></ref>
<ref id="ref44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname><given-names>S.</given-names></name> <name><surname>Li</surname><given-names>X.</given-names></name> <name><surname>Chen</surname><given-names>Y.</given-names></name> <name><surname>Chandrasekaran</surname><given-names>S. T.</given-names></name> <name><surname>Sanyal</surname><given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>Temporal-coded deep spiking neural network with easy training and robust performance</article-title>. <source>AAAI</source> <volume>35</volume>, <fpage>11143</fpage>&#x2013;<lpage>11151</lpage>. doi: <pub-id pub-id-type="doi">10.1609/aaai.v35i12.17329</pub-id></citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname><given-names>C.</given-names></name> <name><surname>Cui</surname><given-names>X.</given-names></name> <name><surname>Feng</surname><given-names>S.</given-names></name> <name><surname>Chen</surname><given-names>G.</given-names></name> <name><surname>Zhong</surname><given-names>Y.</given-names></name> <name><surname>Dai</surname><given-names>Z.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>An all integer-based spiking neural network with dynamic threshold adaptation</article-title>. <source>Front. Neurosci.</source> <volume>18</volume>:<fpage>20</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fnins.2024.1449020</pub-id>, PMID: <pub-id pub-id-type="pmid">39741532</pub-id></citation></ref>
</ref-list>
</back>
</article>