<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2025.1512926</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Fine spatial-temporal density mapping with optimized approaches for many-core system</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes" equal-contrib="yes">
<name><surname>Wang</surname> <given-names>Song</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2871388/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Gao</surname> <given-names>Yiyuan</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2161682/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Seng</surname> <given-names>Bingfeng</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Pei</surname> <given-names>Jing</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/400576/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Yuan</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Huang</surname> <given-names>Jianqiang</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Computer Technology and Application, Qinghai University</institution>, <addr-line>Xining</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Qinghai Provincial Laboratory for Intelligent Computing and Application, Qinghai University</institution>, <addr-line>Xining</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Precision Instrument, Center for Brain Inspired Computing Research (CBICR), Tsinghua University</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Qinghai Provincial Green Computing Power Engineering Technology Research Center</institution>, <addr-line>Xining</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Amirreza Yousefzadeh, University of Twente, Netherlands</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Chenglong Zou, Peking University, China</p>
<p>Liliana Ibeth Barbosa Santillan, University of Guadalajara, Mexico</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Song Wang <email>2024990013&#x00040;qhu.edu.cn</email></corresp>
<fn fn-type="equal" id="fn001"><p>&#x02020;These authors have contributed equally to this work and share first authorship</p></fn></author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>04</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>19</volume>
<elocation-id>1512926</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>20</day>
<month>03</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Wang, Gao, Seng, Pei, Zhang and Huang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wang, Gao, Seng, Pei, Zhang and Huang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>A fine mapping strategy is essential for optimizing the layout and execution speed of large-scale neural networks on many-core systems. However, the benefits of many-core systems diminish when applied to neural networks with significant data and computational demands, due to imbalanced resource utilization between space and time when relying on existing single spatial or temporal mapping strategies. To tackle this challenge, we introduce the concept of spatial-temporal density and propose a spatial-temporal density mapping method to fully leverage both spatial and computational resources. Within the framework of the proposed method, we further introduce two approaches: the Negative Sequence Memory Management (NSM) method, which enhances spatial resource (i.e. core memory) utilization, and the Many-core Parallel Synchronous (MPS) approach, which optimizes computational resource (i.e. core multiply and accumulate units, MACs) utilization. To demonstrate the superiority of these methods, the mapping techniques are implemented on our state-of-the-art many-core chip, TianjicX. The results indicate that the NSM method improves spatial utilization by a factor of 3.05 compared to the traditional Positive Sequence Memory Management (PSM) method. Furthermore, the MPS approach increases computational speed by 6.7% relative to the previously widely adopted pipelined method. Overall, the spatial-temporal density mapping method improves system performance by a factor of 1.85 compared to the commonly employed layer-wise mapping method, effectively balancing spatial and temporal resource utilization.</p></abstract>
<kwd-group>
<kwd>many-core</kwd>
<kwd>spatial-temporal density mapping</kwd>
<kwd>memory management</kwd>
<kwd>spatial resource</kwd>
<kwd>computational speed</kwd>
</kwd-group>
<counts>
<fig-count count="12"/>
<table-count count="3"/>
<equation-count count="13"/>
<ref-count count="53"/>
<page-count count="14"/>
<word-count count="8112"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Neuromorphic Engineering</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Recent many-core architectures have been widely adopted by accelerators (Shao et al., <xref ref-type="bibr" rid="B33">2019</xref>; Chen et al., <xref ref-type="bibr" rid="B6">2019</xref>; Modha et al., <xref ref-type="bibr" rid="B26">2023</xref>) and neuromorphic chips (Sawada et al., <xref ref-type="bibr" rid="B32">2016</xref>; Shen et al., <xref ref-type="bibr" rid="B34">2016</xref>; Benjamin et al., <xref ref-type="bibr" rid="B3">2014</xref>; Pei et al., <xref ref-type="bibr" rid="B29">2019</xref>; Ma et al., <xref ref-type="bibr" rid="B25">2022</xref>; Davies et al., <xref ref-type="bibr" rid="B8">2018</xref>; Shrestha et al., <xref ref-type="bibr" rid="B35">2024</xref>; Ambrogio et al., <xref ref-type="bibr" rid="B1">2023</xref>; Le Gallo et al., <xref ref-type="bibr" rid="B22">2023</xref>) due to their low power consumption and high parallelism. A crucial aspect of many-core systems involves mapping neural networks into pipeline groups, where each group is assigned a cluster of cores to handle computational tasks. In a many-core system, spatial resources correspond to the memory storage capacity of each core, which is closely associated with the number of parameters in a neural network. Computational resources refer to the number of multipliers and accumulators in each core, which are closely related to the computational workload of the neural network. In homogeneous many-core systems, the temporal and computational resources are consistent across cores. However, the distribution of parameters and computational workload between the layers of a neural network is imbalanced. Two common strategies to implement neural networks are temporal mapping and spatial mapping, as illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>. In temporal mapping, tasks are executed through time slicing, with each Processing Element (PE) or core independently accessing data and taking on tasks according to their time complexity. However, this approach leads to data duplication between cores, resulting in inefficient utilization of spatial resources. Although this method eliminates tail latency between cores, it incurs significant data movement between cores and external storage due to the limited memory capacity of the cores, as shown on the left side of <xref ref-type="fig" rid="F1">Figure 1</xref>. Alternatively, spatial mapping divides tasks according to spatial slicing, where clusters of cores are assigned tasks based on spatial volume (Ma et al., <xref ref-type="bibr" rid="B25">2022</xref>). Although this method reduces data movement, it introduces tail latency across clusters, leading to inefficient utilization of computational (i.e., temporal) resources, as shown on the right side of <xref ref-type="fig" rid="F1">Figure 1</xref>. Moreover, the layer-wise mapping approach, commonly employed in many-core systems (Zimmer et al., <xref ref-type="bibr" rid="B52">2020</xref>; Chen et al., <xref ref-type="bibr" rid="B6">2019</xref>; Pei et al., <xref ref-type="bibr" rid="B29">2019</xref>; Le Gallo et al., <xref ref-type="bibr" rid="B22">2023</xref>), integrates both temporal and spatial mapping strategies, offering improved mapping efficiency compared to single-method approaches. However, it still encounters issues of imbalance between spatial and computational resources, as inconsistencies in the partitioning scheme across layers may introduce latency during reshaping between adjacent layers. To tackle this challenge, we propose spatial-temporal density mapping to balance the utilization between the spatial resource and the computational resource.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Illustration of the distinction between temporal mapping and spatial mapping scheme: the left of the figure shows that a large amount of data accesses the external memory in the temporal mapping, the right of the figure shows that the tail-latency exists in the spatial mapping which is determined by the longest time.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0001.tif"/>
</fig>
<p>Nevertheless, the spatial-temporal mapping scheme faces challenges like memory constraints and time delays. During the mapping process, spatial resources are prioritized when allocating cores. Because core memory space directly impacts data movement and memory access, which play a critical role in chip energy consumption and memory footprints (Han et al., <xref ref-type="bibr" rid="B13">2016</xref>; Chen et al., <xref ref-type="bibr" rid="B5">2016</xref>). In neural networks, most historical and intermediate data must either be discarded or updated during computation (Hu et al., <xref ref-type="bibr" rid="B15">2021</xref>), allowing memory space to be reused once it is freed. Current research efforts have largely focused on reducing memory footprints and data movement in traditional hardware systems. Techniques such as fine-grained memory management (Nie et al., <xref ref-type="bibr" rid="B28">2022</xref>), reinforcement-based memory management for GPUs (Liu et al., <xref ref-type="bibr" rid="B24">2021</xref>), and machine intelligence-driven hybrid memory management (Doudali and Gavrilovska, <xref ref-type="bibr" rid="B11">2022</xref>) have shown promising results. However, there remains a notable gap in research addressing storage management strategies tailored specifically for many-core systems. When mapping large neural networks onto hardware, partitioning is necessary due to the limited memory and MACs available on individual cores. In practice, the input channel (<italic>C</italic><sub><italic>in</italic></sub>) of the neural network is typically selected for partitioning, as the channel dimension is strongly correlated with the computational load, including multiplications and accumulations (Xie et al., <xref ref-type="bibr" rid="B46">2017</xref>). Partitioning along the <italic>C</italic><sub><italic>in</italic></sub> dimension results in the generation of partial sums (Psums) across multiple cores. To ensure computational precision, hardware architectures such as TianjicX (Pei et al., <xref ref-type="bibr" rid="B29">2019</xref>) and Simba (Shao et al., <xref ref-type="bibr" rid="B33">2019</xref>) expand the bit-width of these Psums. However, the increased width of Psums necessitates their accumulation across cores or PEs, leading to significant communication latency. Given these two aspects, it is critical to develop optimized methods for memory management and Psums computation to improve the utilization of both spatial and computational resources, thus enhancing memory efficiency and reducing computational latency.</p>
<p>In this work, we propose a fine-grained spatial-temporal density mapping scheme to balance the utilization of spatial and computational resources. Decentralized many-core systems (Lin et al., <xref ref-type="bibr" rid="B23">2018</xref>; Shen et al., <xref ref-type="bibr" rid="B34">2016</xref>; Benjamin et al., <xref ref-type="bibr" rid="B3">2014</xref>; Pei et al., <xref ref-type="bibr" rid="B29">2019</xref>; Ma et al., <xref ref-type="bibr" rid="B25">2022</xref>; Zhong et al., <xref ref-type="bibr" rid="B51">2024</xref>) are well-suited for executing multiple neural networks concurrently, enabling the simultaneous exploitation of both spatial and temporal complexities. To demonstrate the feasibility of our proposed mapping scheme, we leverage our state-of-the-art many-core chip, TianjicX (Ma et al., <xref ref-type="bibr" rid="B25">2022</xref>). TianjicX chip is capable of spatial-temporal elasticity, effectively executing and coordinating multiple tasks in parallel. Various neural network models have been successfully deployed on TianjicX chip (Zheng et al., <xref ref-type="bibr" rid="B50">2024</xref>; Wu et al., <xref ref-type="bibr" rid="B44">2024</xref>), which has also been utilized for applications such as gaming and place recognition in edge robotics (Ma et al., <xref ref-type="bibr" rid="B25">2022</xref>; Yu et al., <xref ref-type="bibr" rid="B48">2023</xref>). To further explore the advantages of spatial-temporal density mapping, we introduce Negative Sequence Memory Management (NSM) method to enhance spatial resource utilization, and Many-core Parallel Synchronous (MPS) approach to optimize computational resource utilization. We conduct a theoretical analysis of the proposed spatial-temporal density mapping scheme and applied it to map the typical neural network architecture, ResNet-50, onto TianjicX hardware. Experimental results show that our method outperforms traditional mapping approaches, demonstrating its superiority in terms of efficiency.</p>
<p>The remainder of this article is organized as follows: Section 2 provides background on spatial and temporal mapping, along with memory space management and partial sum computation. Section 3 details the proposed approach, including the theoretical analysis of spatial-temporal density mapping, NSM, and MPS. The hardware implementation of TianjicX is discussed in Section 4. Section 5 presents the experimental results of the proposed mapping method, followed by a comparative analysis with traditional mapping approaches. Finally, Section 6 concludes the paper and offers insights for future work.</p>
</sec>
<sec id="s2">
<title>2 Related works</title>
<sec>
<title>2.1 Spatial and temporal mapping</title>
<p>The hardware architecture dictates the mapping scheme. Neuromorphic chip architectures based on crossbar arrays employ mapping schemes to enhance memory space utilization (Amir et al., <xref ref-type="bibr" rid="B2">2013</xref>; Cui et al., <xref ref-type="bibr" rid="B7">2022</xref>; Wei et al., <xref ref-type="bibr" rid="B42">2022</xref>; Rueckauer et al., <xref ref-type="bibr" rid="B31">2022</xref>; Zou et al., <xref ref-type="bibr" rid="B53">2021</xref>). For instance, a compiler has been proposed that utilizes a greedy layer-wise optimization algorithm and connection sharing to minimize the duplication of weight kernels in convolutional topologies for the Loihi core (Davies et al., <xref ref-type="bibr" rid="B8">2018</xref>; Rueckauer et al., <xref ref-type="bibr" rid="B31">2022</xref>). This approach achieved near-optimal space resource utilization of 80 % across 16 chips for a 28-layer network. Similarly, FangTianSim has been introduced, which flattens input images and output neurons into one-dimensional arrays for mapping spiking neural networks (Wei et al., <xref ref-type="bibr" rid="B42">2022</xref>). This method aims to improve the utilization of resistive random-access memory (RRAM) in crossbar structures. Additionally, channel-major search and square-major search algorithms have been proposed to ensure high resource efficiency and compactness in hardware modules (Zou et al., <xref ref-type="bibr" rid="B53">2021</xref>). These algorithms also introduced density metrics for axons, neurons, and synapses as practical evaluation criteria for assessing crossbar resource efficiency. In addition to spatial mapping, some neuromorphic chips employ temporal mapping schemes. A loop representation with simulated annealing has been used to place local structures on hardware, minimizing communication hops and optimizing energy costs (Cui et al., <xref ref-type="bibr" rid="B7">2022</xref>). Another approach involves a fully-unfolded temporal mapping that reuses limited crossbar resources through time-division multiplexing (Esser et al., <xref ref-type="bibr" rid="B12">2016</xref>). However, this method incurs substantial memory overhead despite achieving high computational parallelism. To balance spatial and temporal resource utilization, a semi-folded mapping paradigm has been proposed that strategically allocates resources to optimize overall efficiency (Deng et al., <xref ref-type="bibr" rid="B10">2018</xref>). In a recent study, a neuromorphic chip with 1024 scalable cores was designed, and mapping experiments demonstrated that the semi-folded mapping strategy significantly reduced core overhead by a factor of 0.07, while only moderately decreasing inference throughput by a factor of 0.04 (Zhong et al., <xref ref-type="bibr" rid="B51">2024</xref>). Despite these advancements, the challenge of balancing spatial and temporal resource utilization in crossbar-based neuromorphic architectures remains a critical issue. While substantial research has focused on improving spatial resource utilization through various mapping schemes, the exploration of temporal efficiency in artificial neural networks (ANNs) is still relatively limited. Similarly, mapping ANNs onto many-core architectures also faces the challenge of unbalanced utilization efficiency between spatial resources and computational resources. However, research on effectively balancing spatial-temporal resource utilization for ANNs remains underexplored to date.</p>
<p>On the other hand, most many-core designed without crossbar achitecture adopts layer-wise approach for artificial neural network mapping (Chen et al., <xref ref-type="bibr" rid="B6">2019</xref>; Pei et al., <xref ref-type="bibr" rid="B29">2019</xref>; Le Gallo et al., <xref ref-type="bibr" rid="B22">2023</xref>; Zimmer et al., <xref ref-type="bibr" rid="B52">2020</xref>), as illustrated in <xref ref-type="fig" rid="F2">Figures 2a</xref>, <xref ref-type="fig" rid="F2">b</xref> in a pipelined manner. The distribution of parameters and computational workload across each layer is imbalanced, leading to tail latency being determined by the longest execution time among the allocated cluster cores. While TianjicX architecture (Ma et al., <xref ref-type="bibr" rid="B25">2022</xref>) represents a pioneering effort in integrating spatial-temporal mapping for multi-task processing, its underlying spatial-temporal coordination mechanism has not been thoroughly explored, particularly in terms of strategies for mitigating tail latency. To address the critical challenge of optimizing spatial-temporal resource utilization in ANNs, this paper proposes a novel spatial-temporal density mapping framework.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The process of generating partial sum: <bold>(a)</bold> neural network with multi-layers; <bold>(b)</bold> layer-wise mapping by temporal mapping or spatial mapping; <bold>(c)</bold> partitioning the neural networks; <bold>(d)</bold> process of resulting Psums.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0002.tif"/>
</fig>
</sec>
<sec>
<title>2.2 Management of the memory space</title>
<p>Efficient memory management systems are crucial for minimizing memory footprint and reducing data movement. Existing memory management systems are predominantly designed for the Von Neumann architecture, which relies on external memory. The vDNN architecture (Rhu et al., <xref ref-type="bibr" rid="B30">2016</xref>) introduces a swap strategy and employs a layer-wise memory management approach. Several studies (Huang et al., <xref ref-type="bibr" rid="B16">2020</xref>; Jiang et al., <xref ref-type="bibr" rid="B17">2019</xref>; Wahib et al., <xref ref-type="bibr" rid="B39">2020</xref>) address GPU memory footprint reduction using coarse-grained methods. These methods typically involve swapping or recomputing data during the backward phase and evicting tensors during the forward phase. Many of these approaches (Huang et al., <xref ref-type="bibr" rid="B16">2020</xref>; Chen et al., <xref ref-type="bibr" rid="B4">2018</xref>; Xiao et al., <xref ref-type="bibr" rid="B45">2020</xref>) focus on tensor-wise memory management. However, this approach limits the flexibility of the swapping policy (Nie et al., <xref ref-type="bibr" rid="B28">2022</xref>). To overcome this limitation, a fine-grained memory management system based on tensor splitting has been proposed, aiming to alleviate memory bottlenecks while maintaining neural network training efficiency (Nie et al., <xref ref-type="bibr" rid="B28">2022</xref>).</p>
<p>Additionally, various other memory management techniques have been developed, such as reinforcement-based methods for class-incremental learning, holistic approaches for GPU systems, and layer-conscious memory management frameworks for FPGA-based accelerators. Despite these advancements, many-core systems, unlike the Von Neumann architecture, typically lack external memory. As a result, methods for memory management in decentralized many-core systems remain underexplored.</p>
<p>TianjicX chip, a many-core system, utilizes Positive Sequence Memory Management (PSM), as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. However, this approach struggles with efficient memory space reuse. For instance, the space of <italic>S</italic><sub>3</sub> is larger than the combined space of <italic>S</italic><sub>1</sub> and <italic>S</italic><sub>2</sub>, so the <italic>S</italic><sub>2</sub> space can only be reused when both <italic>S</italic><sub>1</sub> and <italic>S</italic><sub>2</sub> spaces are released. To address these limitations and improve memory utilization in many-core systems, we propose a Negative Sequence Memory Management (NSM) approach.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Positive sequence memory management for multi-layers. The data and the start address of each layer are stored along the direction of address increase. The <italic>S</italic><sub>1</sub> has a forward propagation due to the residual connection. And the space of the <italic>S</italic><sub>1</sub> can be released after the phase of the <italic>T</italic><sub>4</sub>. The <italic>S</italic><sub>2</sub> can not be reused in the phase <italic>T</italic><sub>3</sub> in real time for the space of the <italic>S</italic><sub>3</sub> is larger than that of the <italic>S</italic><sub>2</sub>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0003.tif"/>
</fig>
</sec>
<sec>
<title>2.3 Partial sum computing</title>
<p>As shown in <xref ref-type="fig" rid="F2">Figures 2c</xref>, <xref ref-type="fig" rid="F2">d</xref>, partial sums (Psums) are prevalent in the AI hardware computation process, particularly when the <italic>C</italic><sub><italic>in</italic></sub> dimension is selected for partitioning to address limited memory capacity (Shao et al., <xref ref-type="bibr" rid="B33">2019</xref>; Wang et al., <xref ref-type="bibr" rid="B40">2021</xref>; Wu et al., <xref ref-type="bibr" rid="B43">2020</xref>). To reduce both the number of parameters and computations, some researchers have adopted grouped convolution (Xie et al., <xref ref-type="bibr" rid="B46">2017</xref>; Howard et al., <xref ref-type="bibr" rid="B14">2017</xref>; Zhang et al., <xref ref-type="bibr" rid="B49">2019</xref>; Wang et al., <xref ref-type="bibr" rid="B41">2019</xref>), where Psums are directly activated on each core of the GPU platform without aggregation. However, this approach entails a quantifiable trade-off in accuracy (Howard et al., <xref ref-type="bibr" rid="B14">2017</xref>). Empirical validation reveals a 1% reduction in ImageNet classification accuracy when employing direct activation of depth-wise separable convolutions, as opposed to full convolutions where partial sums (Psums) are activated post-aggregation.</p>
<p>To maintain accuracy, most accelerators perform activation after Psums aggregation and expand the bit-width of Psums. To address the challenges posed by large-bit-width Psums, several accelerators adopt a Pipelined Manner (PM) (Shao et al., <xref ref-type="bibr" rid="B33">2019</xref>; Chen et al., <xref ref-type="bibr" rid="B6">2019</xref>; Jouppi et al., <xref ref-type="bibr" rid="B19">2017</xref>; Sze et al., <xref ref-type="bibr" rid="B38">2017</xref>; Yin et al., <xref ref-type="bibr" rid="B47">2017</xref>; Kung et al., <xref ref-type="bibr" rid="B20">2019</xref>). In this approach, Psums are propagated through the processing element (PE) array or cores (Deng et al., <xref ref-type="bibr" rid="B9">2020</xref>) during convolution or Matrix-Vector Multiplication (MVM) operations, which enhances data reuse and reduces the need for memory bandwidth.</p>
<p>TianjicX neuromorphic chip (Pei et al., <xref ref-type="bibr" rid="B29">2019</xref>; Wang et al., <xref ref-type="bibr" rid="B40">2021</xref>) adopts cluster cores with a dedicated function for managing Psums through Vector-Vector Accumulation (VVA). However, this configuration introduces latency between the VVA cores and other cores within TianjicX chip. To mitigate the latency associated with computing Psums, we propose the Many-core Parallel Synchronous (MPS) approach for Psums computation.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Motivation and approach</title>
<sec>
<title>3.1 Spatial-temporal density mapping</title>
<p>TianjicX chip can support flexible mapping schemes, such as temporal mapping, spatial mapping, and spatial-temporal mapping. Firstly, we define a concept of spatial-temporal density (&#x003C1;) for a computing core. The spatial-temporal density can be described as the Calculation Amounts (CA) per unit space (S) of a core. The spatial-temporal density of a core can be described by <xref ref-type="disp-formula" rid="E1">Equation 1</xref>:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>C</mml:mi><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the <italic>i</italic> represents the <italic>i</italic>-th core. Assuming <italic>N</italic> tasks are assigned to <italic>M</italic> cores, it can be described by <xref ref-type="disp-formula" rid="E2">Equation 2</xref>:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>C</mml:mi><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>O</mml:mi><mml:mi>P</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the OPs represent the operations of the task. Therefore, the variance of the core density &#x003C3;<sub>&#x003C1;</sub> can be described as follows <xref ref-type="disp-formula" rid="E3">Equation 3</xref>:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>As shown in <xref ref-type="fig" rid="F4">Figure 4a</xref>, the density of each cluster core varies under spatial mapping. <xref ref-type="fig" rid="F4">Figure 4b</xref> illustrates that the mapping scheme may fail if the first task occupies the largest memory space, requiring the allocation of 9 cores. While temporal mapping reduces latency, it leads to inefficient memory utilization. In contrast, spatial-temporal mapping effectively leverages the core density. The Tianjicat (Ma et al., <xref ref-type="bibr" rid="B25">2022</xref>) adopts a coarse density mapping scheme, as depicted in <xref ref-type="fig" rid="F4">Figure 4c</xref>.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>TianjicX supports flexible mapping. <bold>(a)</bold> Spatial mapping allocates cores according to the parameters of the task; <bold>(b)</bold> temporal mapping allocates cores according to the operations of the task; <bold>(c)</bold> coarse spatial-temporal density mapping allocates cores according to both the parameters and operations naively; <bold>(d)</bold> fine spatial-temporal density mapping.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0004.tif"/>
</fig>
<p>To further optimize spatial-temporal density, we propose a fine spatial-temporal density mapping strategy. As shown in <xref ref-type="fig" rid="F4">Figure 4d</xref>, this approach evenly distributes the operations and parameters of each layer across 16 cores. Notably, in this configuration, the density &#x003C1; of each core is uniform, and the standard deviation &#x003C3;<sub>&#x003C1;</sub> is zero, as observed in <xref ref-type="fig" rid="F4">Figure 4d</xref>.</p>
<p>In the fine spatial-temporal density mapping scheme, the output activation for each layer does not require communication, since the partitioning of each layer is consistent and the output activation is stored locally. Consequently, reshaping latency caused by partition mismatches and communication latency between adjacent layers are eliminated. Moreover, by reusing the space of the preceding layer&#x00027;s input activation, memory consumption is further reduced.</p>
</sec>
<sec>
<title>3.2 Memory management for many-core system</title>
<p>In deep neural networks, the input to a given layer is the output of the preceding layer, and pipelined execution occurs between adjacent layers. Since activation data must be continuously updated, the memory allocated for output activations can be dynamically reused for input activations in real time, provided there is no data scrambling. To enhance memory reuse efficiency, we propose a negative sequence memory management (NSM) strategy for multi-layer execution, as illustrated in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Negative sequence memory management for multi-layers. The data is stored along the direction of address increasing. While the start address of each layer is stored along the direction of the address decreasing. The space of the post layer can be reused as that of the current layer in real-time.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0005.tif"/>
</fig>
<p>As shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, the memory allocated to <italic>S</italic><sub>1</sub> is not immediately reused during forward propagation. However, the memory regions allocated to <italic>S</italic><sub>2</sub> and <italic>S</italic><sub>4</sub> can be efficiently reused in real time through the NSM mechanism. Once <italic>S</italic><sub>1</sub> completes its forward propagation and is released, its memory can also be reused. Compared to conventional positive sequence memory management, NSM effectively reduces the peak memory footprint, which is determined by the combined memory usage of <italic>S</italic><sub>1</sub> and <italic>S</italic><sub>3</sub>, leading to improved memory efficiency.</p>
<p>In order to avoid a scrambling between the input activation and output activation in real-time, the relationship of start address between two adjacent layers can be described as follows:</p>
<p>if (<italic>V</italic><sub><italic>i</italic>&#x0002B;1</sub> &#x0003E; <italic>V</italic><sub><italic>i</italic></sub>)</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>d</mml:mi><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>A</mml:mi><mml:mi>d</mml:mi><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>else</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>d</mml:mi><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>A</mml:mi><mml:mi>d</mml:mi><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the <italic>Addr</italic><sub><italic>i</italic></sub> represents the address of input activation or intermediate data in the <italic>i</italic>-th layer, the <italic>V</italic><sub><italic>i</italic>&#x0002B;1</sub> represents the volume of the space parameters, the <italic>const</italic> is an address constant which is determined by the hardware. The scrambling of space between the adjacent layers can be eliminated by adjusting the value of <italic>const</italic>.</p>
</sec>
<sec>
<title>3.3 Many-core Parallel Synchronous (MPS) computing partial sum</title>
<p>After allocating spatial-temporal resources to each cluster core, TianjicX system proceeds with neural network computation. As previously discussed, Psums must be efficiently managed during this process. TianjicX supports multiple approaches for addressing Psums, including the Step-by-Step (SS) method, the Dichotomy Step-by-Step (DSS) method, and the Many-core Parallel Synchronous (MPS) method. When using a <italic>C</italic><sub><italic>in</italic></sub> partitioning scheme with <italic>M</italic> &#x0003D; 4 groups, these methods are illustrated in <xref ref-type="fig" rid="F6">Figure 6</xref>. Notably, the SS and DSS methods operate asynchronously, leading to inefficient utilization of computational resources, as some cores remain idle during Psums processing.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Illustration of supporting flexible methods for computing Psums: <bold>(a)</bold> Step-by-Step (SS) method; <bold>(b)</bold> Dichotomy Step-by-Step (DSS) method proposed by wallace tree; <bold>(c)</bold> Many-core Parallel Synchronous (MPS) method.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0006.tif"/>
</fig>
<p>To achieve higher computational efficiency and minimize resource wastage, we adopt the MPS method. The detailed process is illustrated in <xref ref-type="fig" rid="F6">Figure 6c</xref>. In this approach, each core partitions its assigned Psums into <italic>M</italic> groups, corresponding to the number of <italic>C</italic><sub><italic>in</italic></sub> partitioning groups. This process can be formally expressed as <xref ref-type="disp-formula" rid="E6">Equation 6</xref>.</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>&#x02200;</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x02260;</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msubsup><mml:mrow><mml:mo>&#x0222A;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x02229;</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x02205;</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>n</italic> represents the <italic>n</italic>-th core. Secondly, the partitioning Psums sets of each core are communicated to other cores concurrently based on their corresponding serial numbers. Finally, each core aggregates Psums synchronously after communication. The aggregation and results of each core can be described as <xref ref-type="disp-formula" rid="E7">Equations 7</xref>, <xref ref-type="disp-formula" rid="E8">8</xref>.</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>&#x02200;</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mi>O</mml:mi><mml:mi>A</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E8"><label>(8)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>&#x02200;</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x02260;</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msubsup><mml:mrow><mml:mo>&#x0222A;</mml:mo></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup><mml:mi>O</mml:mi><mml:mi>A</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>O</mml:mi><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>O</mml:mi><mml:mi>A</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x02229;</mml:mo><mml:mi>O</mml:mi><mml:mi>A</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x02205;</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the <italic>OA</italic>[<italic>n</italic>] represents the <italic>n</italic>-th set of the output activation. By the method of MPS, each core can compute 1/<italic>M</italic> parts of output activation set concurrently. Throughout the entire addressing Psums process, all the cores work in communication and computation synchronously all the time. Therefore, the computation speed of MPS is faster than that of SS or DSS.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Implementation details</title>
<p>TianjicX chip was fabricated using the UMC 28-nm High Performance Compact Plus (HPC&#x0002B;) CMOS process and assembled in an FPGA-225 package. Prior studies have demonstrated its outstanding performance in terms of fundamental characteristics, computational capabilities, and power efficiency (Ma et al., <xref ref-type="bibr" rid="B25">2022</xref>). TianjicX employs a decentralized many-core architecture comprising 160 functional cores, as shown in <xref ref-type="fig" rid="F7">Figure 7</xref>. Additionally, the chip supports flexible mapping strategies and partitioning methods, allowing for optimized neural network execution. TianjicX architecture employs a fully digital design featuring a non-crossbar memory through innovative memory addressing schemes. Each computational core functions as a reconfigurable processing engine capable of executing vector-matrix multiplication (VMM), vector-vector multiplication (VVM), and vector-vector accumulation (VVA) operations through synergistic collaboration of its six functional modules: controller, axon, dendrite, dual-port memory (2 &#x000D7; 64KB), soma, and router. The axon module specializes in the orchestration of tensor data and input buffering for dendrite operations, while preprocessed input vectors and synaptic weights are managed by an external controller and stored in the dual memory banks of the core. The dendrite module incorporates a high-throughput arithmetic unit with 128 parallel 8-bit multipliers coupled with 128 32-bit signed accumulators, enabling simultaneous multiply and accumulate (MAC) operations. Post-processing operations including nonlinear activation functions (ReLU), leakage integration mechanisms, and spiking neural models (LIF) are implemented in the soma module through configurable data transformation pipelines. The axon and soma operations employed in the subsequent experiments are listed in <xref ref-type="table" rid="T1">Table 1</xref>. The router is responsible for data communication between each core. The architecture implements unified memory addressing with dynamic resource allocation, allowing flexible memory partitioning and shared access across computational modules. This memory virtualization scheme supports various neural network paradigms through software-defined memory mapping, enabling efficient execution of neural networks.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>The architecture of the TianjicX based on a many-core design implemented with digital circuits.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0007.tif"/>
</fig>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Axon and soma operations.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>&#x00023; Uint</bold></th>
<th valign="top" align="center"><bold>Operation</bold></th>
<th valign="top" align="center"><bold>Defination</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Axon</td>
<td valign="top" align="center">VMM</td>
<td valign="top" align="center">y = w &#x000B7; x</td>
</tr> <tr>
<td valign="top" align="left">Axon</td>
<td valign="top" align="center">VVA</td>
<td valign="top" align="center">y = <inline-formula><mml:math id="M9"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula></td>
</tr> <tr>
<td valign="top" align="left">Soma</td>
<td valign="top" align="center">Relu</td>
<td valign="top" align="center">y = max(x,0)</td>
</tr></tbody>
</table>
</table-wrap>
<p>The experimental setup includes an Intel Arria 10 FPGA, a host computer, TianjicX chip, and an oscilloscope, as depicted in <xref ref-type="fig" rid="F8">Figure 8</xref>. Neural network parameters and inputs are configured and downloaded onto the chip via dedicated software on the host computer. Execution time is measured using a RIGOL MSO8104 oscilloscope. Prior to deployment on hardware, experiments are first simulated using TianjicX simulator, which faithfully replicates the real chip&#x00027;s behavior. This simulation system employs a dual-verification mechanism: the Behavioral Simulator and the Cycle-Accurate Simulator, which are implemented in different programming languages. The simulation results are deemed reliable only when the outputs from both levels of simulators are completely consistent. Once validated in simulation, the corresponding configuration files are downloaded to the chip for final execution.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Testing system based on TianjicX neuromorphic chip.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0008.tif"/>
</fig>
<p>The ResNet-50 is often adopted to benchmark by many hardwares (Myung et al., <xref ref-type="bibr" rid="B27">2021</xref>; Zimmer et al., <xref ref-type="bibr" rid="B52">2020</xref>; Jouppi et al., <xref ref-type="bibr" rid="B18">2021</xref>), which contains basic operator of Conv, Pooling, Skip and Relu. First, to evaluate the effectiveness of the proposed negative sequence memory management, all blocks of ResNet-50 are mapped onto the cores of TianjicX. Second, to assess the performance improvements achieved by the Many-core Parallel Synchronous (MPS) method, we conduct simulations comparing SS, DSS, PM, and MPS approaches. Finally, to demonstrate the benefits of fine spatial-temporal density mapping, we implement two neural network configurations on TianjicX chip: (1) a network consisting of two blocks (2b, 2c) from ResNet-50 and (2) a custom-designed network derived from the first configuration by removing residual connections. The parameters of the designed network are summarized in <xref ref-type="fig" rid="F9">Figure 9</xref>.</p>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>A part blocks of ResNet-50 network and a custom-designed network.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0009.tif"/>
</fig>
</sec>
<sec id="s5">
<title>5 Experiments and analysis</title>
<sec>
<title>5.1 Simulation of the memory management</title>
<p>To evaluate the space utilization of the negative sequence memory management mechanism, all blocks of ResNet-50 are partitioned and mapped onto TianjicX cores. Both negative sequence and positive sequence memory management approaches are tested separately. The analysis results are presented in <xref ref-type="fig" rid="F10">Figure 10</xref>.</p>
<fig id="F10" position="float">
<label>Figure 10</label>
<caption><p>The space utilization of a core by the negative sequence memory management and the positive sequence memory management.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0010.tif"/>
</fig>
<p>As illustrated in <xref ref-type="fig" rid="F10">Figure 10</xref>, our proposed NSM mechanism exhibits superior memory efficiency when compared to the conventional PSM approach. The experimental results yield three key observations: Firstly, the implementation of PSM for Blocks 2b and 2c leads to a negative residual memory capacity of -14 KB, signifying memory overflow and the necessity for additional core allocation to fully map the network. Secondly, NSM achieves its maximum optimization in Block 4a, where the residual memory capacity reaches 45 KB, marking a 3.75-fold improvement over the 12 KB achieved by PSM. Thirdly, a system-level analysis across all 17 benchmark blocks demonstrates that NSM increases the total available memory from 204 KB with PSM to 631.5 KB, which represents an average enhancement of 3.05 times.</p>
<p>Since NSM enables real-time memory release, it enhances memory utilization compared to PSM. By adopting NSM, input activation and output activation can be updated dynamically, allowing memory to be predominantly allocated for weight storage. Consequently, the weights of multiple tasks can be accommodated within the memory of a single core, further optimizing resource efficiency.</p>
</sec>
<sec>
<title>5.2 Simulation and analysis for the MPS</title>
<p>To evaluate the performance of MPS in computing partial sums (Psums), we conducted a simulation to compare the computational efficiency of various methods. Based on the computation process of Psums, all methods can be divided into two phases: the communication phase and the computation phase. Let <italic>t</italic><sub>1</sub> and <italic>t</italic><sub>2</sub> denote the execution times of the communication phase and computation phase, respectively.</p>
<p>We assume that the number of cores is <italic>m</italic>, which corresponds to the number of groups in the partitioning of <italic>C</italic><sub><italic>in</italic></sub>. The amount of data handled by each core during communication is denoted as <italic>X</italic> (in KB). The hardware processing speed for data communication and data addition is represented by <italic>K</italic><sub>1</sub> (in B/s) and <italic>K</italic><sub>2</sub>, respectively.</p>
<p>By employing 8-bit integers for weights and activations, and 32-bit integers for Psums, the total execution time for the SS, DSS, PM, and MPS methods can be computed as follows:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mfrac><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mfrac><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>X</mml:mi><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E10"><label>(10)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>D</mml:mi><mml:mi>S</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo class="qopname">log</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mfrac><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo class="qopname">log</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mfrac><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo class="qopname">log</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mi>X</mml:mi><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E11"><label>(11)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mi>X</mml:mi><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E12"><label>(12)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>P</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mfrac><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mi>m</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mfrac><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mi>m</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>X</mml:mi><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E13"><label>(13)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>P</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mi>P</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here, <italic>t</italic><sub><italic>SS</italic></sub>, <italic>t</italic><sub><italic>DSS</italic></sub>, <italic>t</italic><sub><italic>PM</italic></sub>, and <italic>t</italic><sub><italic>MPS</italic></sub> represent the execution times under the SS, DSS, PM, and MPS methods, respectively. The simulation results are presented in <xref ref-type="fig" rid="F11">Figure 11</xref>. The SS method is employed by the mixed-signal in-memory computing chip proposed in Le Gallo et al. (<xref ref-type="bibr" rid="B22">2023</xref>). The DSS method, based on the Wallace tree multiplier, has been widely adopted by many designers (Lakshmi et al., <xref ref-type="bibr" rid="B21">2021</xref>; Solanki et al., <xref ref-type="bibr" rid="B36">2021</xref>; Srinivas and Umapathi, <xref ref-type="bibr" rid="B37">2022</xref>). The PM method, on the other hand, is commonly utilized by accelerators (Shao et al., <xref ref-type="bibr" rid="B33">2019</xref>; Chen et al., <xref ref-type="bibr" rid="B6">2019</xref>; Jouppi et al., <xref ref-type="bibr" rid="B19">2017</xref>; Sze et al., <xref ref-type="bibr" rid="B38">2017</xref>; Yin et al., <xref ref-type="bibr" rid="B47">2017</xref>; Kung et al., <xref ref-type="bibr" rid="B20">2019</xref>).</p>
<fig id="F11" position="float">
<label>Figure 11</label>
<caption><p>The running time of the methods for processing the Psums.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0011.tif"/>
</fig>
<p>From the simulation results, it is evident that the MPS method achieves the shortest runtime compared to the other approaches. In particular, the MPS method, implemented by TianjicX platform, demonstrates significant superiority in enhancing the computation speed of Psums compared to the PM method, which is widely adopted by many accelerators. Specifically, the number of processing elements (PEs) in the accelerators is 16, corresponding to a partitioning of <italic>C</italic><sub><italic>in</italic></sub> into 16 groups. MPS demonstrates enhanced runtime performance when utilizing 16 PEs, achieving a reduction in computational latency of 6.7% compared to the PM, as delineated in <xref ref-type="disp-formula" rid="E13">Equation 13</xref>. Furthermore, the speedup factor displays an inverse relationship with the number of PEs, thereby attaining the highest efficiency gains at this configuration.</p>
</sec>
<sec>
<title>5.3 Experiment of spatial-temporal density mapping</title>
<p>The layer-wise mapping approach is widely utilized by accelerators and neuromorphic chips in a pipelined manner (Zimmer et al., <xref ref-type="bibr" rid="B52">2020</xref>; Pei et al., <xref ref-type="bibr" rid="B29">2019</xref>; Le Gallo et al., <xref ref-type="bibr" rid="B22">2023</xref>). Consequently, we use layer-wise mapping as the baseline for comparison. To evaluate the performance of the proposed fine spatial-temporal density mapping scheme, we designed multiple experimental groups with varying spatial-temporal density variances while keeping the total number of allocated cores constant. The allocation of cores and the coupling layer configurations for these groups, corresponding to the two types of networks, are summarized in <xref ref-type="table" rid="T2">Tables 2</xref>, <xref ref-type="table" rid="T3">3</xref>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Different coupling mapping schemes of designed networks.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>&#x00023; cores</bold></th>
<th valign="top" align="center"><bold>Layer-wise</bold></th>
<th valign="top" align="center"><bold>2-layers-c</bold></th>
<th valign="top" align="center"><bold>3-layers-c</bold></th>
<th valign="top" align="center"><bold>6-layers-c</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Layer 1</td>
<td valign="top" align="center">7</td>
<td/>
<td/>
<td/>
</tr> <tr>
<td valign="top" align="left">Layer 2</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">24</td>
<td/>
<td/>
</tr> <tr>
<td valign="top" align="left">Layer 3</td>
<td valign="top" align="center">7</td>
<td/>
<td valign="top" align="center">28</td>
<td/>
</tr> <tr>
<td valign="top" align="left">Layer 4</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">8</td>
<td/>
<td valign="top" align="center">56</td>
</tr> <tr>
<td valign="top" align="left">Layer 5</td>
<td valign="top" align="center">14</td>
<td/>
<td/>
<td/>
</tr> <tr>
<td valign="top" align="left">Layer 6</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">24</td>
<td valign="top" align="center">28</td>
<td/>
</tr> <tr>
<td valign="top" align="left">Aggregation</td>
<td valign="top" align="center">56</td>
<td valign="top" align="center">56</td>
<td valign="top" align="center">56</td>
<td valign="top" align="center">56</td>
</tr> <tr>
<td valign="top" align="left">&#x003C3;<sub>&#x003C1;</sub></td>
<td valign="top" align="center">629.4</td>
<td valign="top" align="center">69.8</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">0</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Different coupling mapping schemes of ResNet-50 (2b,2c).</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>&#x00023; cores</bold></th>
<th valign="top" align="center"><bold>Layer-wise</bold></th>
<th valign="top" align="center"><bold>2-layers-c</bold></th>
<th valign="top" align="center"><bold>3-layers-c</bold></th>
<th valign="top" align="center"><bold>6-layers-c</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Layer 1</td>
<td valign="top" align="center">14</td>
<td/>
<td/>
<td/>
</tr> <tr>
<td valign="top" align="left">Layer 2</td>
<td valign="top" align="center">28</td>
<td valign="top" align="center">42</td>
<td/>
<td/>
</tr> <tr>
<td valign="top" align="left">Layer 3</td>
<td valign="top" align="center">14</td>
<td/>
<td valign="top" align="center">56</td>
<td/>
</tr> <tr>
<td valign="top" align="left">Layer 4</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">28</td>
<td/>
<td valign="top" align="center">112</td>
</tr> <tr>
<td valign="top" align="left">Layer 5</td>
<td valign="top" align="center">28</td>
<td/>
<td/>
<td/>
</tr> <tr>
<td valign="top" align="left">Layer 6</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">42</td>
<td valign="top" align="center">56</td>
<td/>
</tr> <tr>
<td valign="top" align="left">Aggregation</td>
<td valign="top" align="center">112</td>
<td valign="top" align="center">112</td>
<td valign="top" align="center">112</td>
<td valign="top" align="center">112</td>
</tr> <tr>
<td valign="top" align="left">&#x003C3;<sub>&#x003C1;</sub></td>
<td valign="top" align="center">184.1</td>
<td valign="top" align="center">43</td>
<td valign="top" align="center">0</td>
<td valign="top" align="center">0</td>
</tr></tbody>
</table>
</table-wrap>
<p>In <xref ref-type="table" rid="T2">Tables 2</xref>, <xref ref-type="table" rid="T3">3</xref>, the abbreviation x-layers-c denotes the use of <italic>X</italic> coupling layers. By applying the multi-layer coupling method, the spatial-temporal density and the variance &#x003C3;<sub>&#x003C1;</sub> of core utilization can be adjusted. For instance, as shown in <xref ref-type="table" rid="T2">Table 2</xref>, under the layer-wise mapping scheme, the computational tasks for each layer are distributed across 7 cores, 14 cores, and 7 cores, respectively. By systematically varying the spatial-temporal density, we conducted experiments using different mapping schemes, including a 2-layer mapping scheme, a 3-layer mapping scheme, and a 6-layer mapping scheme.</p>
<p>The results, presented in <xref ref-type="fig" rid="F12">Figure 12</xref>, demonstrate that reducing the spatial-temporal density variance &#x003C3;<sub>&#x003C1;</sub> leads to decreases in both execution time and tail latency. In the layer-wise mapping scheme, the computation times for the first and third layers were measured at 259&#x003BC;<italic>s</italic> using an oscilloscope. Similarly, the computation times for the second and fifth layers were recorded at 192&#x003BC;<italic>s</italic>. Notably, the total computation time for the sixth layer, from initiation to completion, was also 259&#x003BC;<italic>s</italic>. In contrast, under the 6-layer coupling scheme illustrated in <xref ref-type="fig" rid="F12">Figure 12a</xref>, where all six layers are interconnected, the computation time for each individual layer was consistently reduced to 140&#x003BC;<italic>s</italic>. All runtime data for these chips were acquired through oscilloscope measurements. <xref ref-type="fig" rid="F12">Figures 12c</xref>, <xref ref-type="fig" rid="F12">d</xref> present two examples, and similar data can be obtained from the provided materials. As a result, the computation speed under the 6-layer coupling scheme was improved by a factor of 1.85 compared to the layer-wise mapping scheme. Furthermore, the tail latency was completely eliminated in the 3-layer coupling and 6-layer coupling mapping schemes.</p>
<fig id="F12" position="float">
<label>Figure 12</label>
<caption><p>The running time and latency of different spatial-temporal densities mapping: <bold>(a)</bold> two blocks of ResNet-50; <bold>(b)</bold> two blocks without residual connection. <bold>(c)</bold> the runtime of chip under the layer-wise mapping tested by the oscilloscope. <bold>(d)</bold> the runtime of chip under the multi-layers-coupling mapping tested by the oscilloscope.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1512926-g0012.tif"/>
</fig>
<p>This improvement is attributed to the fact that no reshaping is required between adjacent layers, as the dimensions of partitions within each layer remain consistent under the 6-layer coupling mapping scheme. Additionally, the output activations in the coupling layers are stored in local memory, eliminating the need for communication between layers. This significantly reduces the proportion of time dedicated to data communication during task execution, thereby enhancing the computational efficiency of the hardware. Since large-scale networks can be divided into multiple flow blocks, this density mapping technique can be broadly applied to full-scale implementations of ResNet-50 and other larger neural networks.</p>
<p>However, it should be noted that full layer-coupling mapping is not universally optimal. Although coupling all layers can effectively eliminate the tail latency of each core by reducing &#x003C3;<sub>&#x003C1;</sub>, it may have adverse effects on computation time and overall latency. As the number of coupled layers increases, the partitioning scheme becomes more complex and less efficient. This leads to increased reshaping latency and a significant reduction in MAC utilization efficiency. Therefore, a trade-off exists between the number of coupled layers and the value of &#x003C3;<sub>&#x003C1;</sub> in spatial-temporal density mapping. This trade-off is ultimately determined by the specific hardware design.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="s6">
<title>6 Conclusion</title>
<p>In this work, we propose spatial-temporal density mapping for the first time that leverages computational resources and spatial resources of a many-core chip. Furthermore, we propose the Negative Sequence Memory Management (NSM) approach to improve space utilization. The NSM can improve space utilization by average 3.05 times compared with the (PSM) used by many-core systems. And we propose the Many-core Parallel Synchronous (MPS) approach to improve the computational speed. It is demonstrated that the MPS can be improved by 6.7% compared to the Pipelined Method (PM) which is adopted by the many-core systems. To demonstrate the superior performance of Spatial-Temporal density Mapping with these optimized approaches, we implement the mapping methods on our state-of-the-art many-core chip, TianjicX. Intensive experiments show that using fine spatial-temporal density mapping improves performance by 1.85x compared to layer-wise mapping used by many-core systems. We believe that optimizing methods for fine spatial-temporal density mapping can help establish a general and efficient mapping framework for many-core systems with variable spatial-temporal density.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>SW: Conceptualization, Funding acquisition, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. YG: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. BS: Data curation, Validation, Writing &#x02013; review &#x00026; editing. JP: Funding acquisition, Project administration, Resources, Supervision, Writing &#x02013; review &#x00026; editing. YZ: Writing &#x02013; review &#x00026; editing. JH: Project administration, Supervision, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported in part by Major science and technology projects in Qinghai Province (2024-GX-A3).</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ambrogio</surname> <given-names>S.</given-names></name> <name><surname>Narayanan</surname> <given-names>P.</given-names></name> <name><surname>Okazaki</surname> <given-names>A.</given-names></name> <name><surname>Fasoli</surname> <given-names>A.</given-names></name> <name><surname>Mackin</surname> <given-names>C.</given-names></name> <name><surname>Hosokawa</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>An analog-ai chip for energy-efficient speech recognition and transcription</article-title>. <source>Nature</source> <volume>620</volume>, <fpage>768</fpage>&#x02013;<lpage>775</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-023-06337-5</pub-id><pub-id pub-id-type="pmid">37612392</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Amir</surname> <given-names>A.</given-names></name> <name><surname>Datta</surname> <given-names>P.</given-names></name> <name><surname>Risk</surname> <given-names>W. P.</given-names></name> <name><surname>Cassidy</surname> <given-names>A. S.</given-names></name> <name><surname>Kusnitz</surname> <given-names>J. A.</given-names></name> <name><surname>Esser</surname> <given-names>S. K.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>&#x0201C;Cognitive computing programming paradigm: a corelet language for composing networks of neurosynaptic cores,&#x0201D;</article-title> in <source>The 2013 International Joint Conference on Neural Networks (IJCNN)</source> (<publisher-loc>Dallas, TX</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>10</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benjamin</surname> <given-names>B. V.</given-names></name> <name><surname>Gao</surname> <given-names>P.</given-names></name> <name><surname>McQuinn</surname> <given-names>E.</given-names></name> <name><surname>Choudhary</surname> <given-names>S.</given-names></name> <name><surname>Chandrasekaran</surname> <given-names>A. R.</given-names></name> <name><surname>Bussat</surname> <given-names>J.-M.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Neurogrid: a mixed-analog-digital multichip system for large-scale neural simulations</article-title>. <source>Proc. IEEE</source> <volume>102</volume>, <fpage>699</fpage>&#x02013;<lpage>716</lpage>. <pub-id pub-id-type="doi">10.1109/JPROC.2014.2313565</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>D. Z.</given-names></name> <name><surname>Hu</surname> <given-names>X. S.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;moDNN: Memory optimal dnn training on gpus,&#x0201D;</article-title> in <source>2018 Design, Automation</source> &#x00026; <italic>Test in Europe Conference</italic> &#x00026; <italic>Exhibition (DATE)</italic> (Dresden: IEEE), <fpage>13</fpage>&#x02013;<lpage>18</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.-H.</given-names></name> <name><surname>Emer</surname> <given-names>J.</given-names></name> <name><surname>Sze</surname> <given-names>V.</given-names></name></person-group> (<year>2016</year>). <article-title>Eyeriss: A spatial architecture for energy-efficient dataflow for convolutional neural networks</article-title>. <source>ACM SIGARCH comp. Architect. News</source> <volume>44</volume>, <fpage>367</fpage>&#x02013;<lpage>379</lpage>. <pub-id pub-id-type="doi">10.1145/3007787.3001177</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.-H.</given-names></name> <name><surname>Yang</surname> <given-names>T.-J.</given-names></name> <name><surname>Emer</surname> <given-names>J.</given-names></name> <name><surname>Sze</surname> <given-names>V.</given-names></name></person-group> (<year>2019</year>). <article-title>Eyeriss v2: A flexible accelerator for emerging deep neural networks on mobile devices</article-title>. <source>IEEE J Emerg. Select. Topics Circuits Syst</source>. <volume>9</volume>, <fpage>292</fpage>&#x02013;<lpage>308</lpage>. <pub-id pub-id-type="doi">10.1109/JETCAS.2019.2910232</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cui</surname> <given-names>X.</given-names></name> <name><surname>Hao</surname> <given-names>X.</given-names></name> <name><surname>Liang</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>G.</given-names></name> <name><surname>Cui</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;A mapping model of snns to neuromorphic hardware,&#x0201D;</article-title> in 2022 IEEE 4th International Conference on Artificial Intelligence Circuits and Systems (AICAS) (<publisher-loc>Incheon</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>206</fpage>&#x02013;<lpage>209</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Davies</surname> <given-names>M.</given-names></name> <name><surname>Srinivasa</surname> <given-names>N.</given-names></name> <name><surname>Lin</surname> <given-names>T.-H.</given-names></name> <name><surname>Chinya</surname> <given-names>G.</given-names></name> <name><surname>Cao</surname> <given-names>Y.</given-names></name> <name><surname>Choday</surname> <given-names>S. H.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Loihi: A neuromorphic manycore processor with on-chip learning</article-title>. <source>IEEE Micro</source> <volume>38</volume>, <fpage>82</fpage>&#x02013;<lpage>99</lpage>. <pub-id pub-id-type="doi">10.1109/MM.2018.112130359</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>G.</given-names></name> <name><surname>Han</surname> <given-names>S.</given-names></name> <name><surname>Shi</surname> <given-names>L.</given-names></name> <name><surname>Xie</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Model compression and hardware acceleration for neural networks: A comprehensive survey</article-title>. <source>Proc. IEEE</source> <volume>108</volume>, <fpage>485</fpage>&#x02013;<lpage>532</lpage>. <pub-id pub-id-type="doi">10.1109/JPROC.2020.2976475</pub-id><pub-id pub-id-type="pmid">38875092</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>L.</given-names></name> <name><surname>Liang</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>G.</given-names></name> <name><surname>Chang</surname> <given-names>L.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name> <name><surname>Ma</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Semimap: A semi-folded convolution mapping for speed-overhead balance on crossbars</article-title>. <source>IEEE Trans. Comp-Aided Design of Integrat. Circ Syst</source>. <volume>39</volume>, <fpage>117</fpage>&#x02013;<lpage>130</lpage>. <pub-id pub-id-type="doi">10.1109/TCAD.2018.2883959</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Doudali</surname> <given-names>T. D.</given-names></name> <name><surname>Gavrilovska</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Toward computer vision-based machine intelligent hybrid memory management,&#x0201D;</article-title> in <source>Proceedings of the International Symposium on Memory Systems, MEMSYS &#x00027;21</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>). <pub-id pub-id-type="doi">10.1145/3488423.3519325</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Esser</surname> <given-names>S. K.</given-names></name> <name><surname>Merolla</surname> <given-names>P. A.</given-names></name> <name><surname>Arthur</surname> <given-names>J. V.</given-names></name> <name><surname>Cassidy</surname> <given-names>A. S.</given-names></name> <name><surname>Appuswamy</surname> <given-names>R.</given-names></name> <name><surname>Andreopoulos</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>From the cover: Convolutional networks for fast, energy-efficient neuromorphic computing</article-title>. <source>Proc. Natl. Acad. Sci. USA</source>. <volume>113</volume>:<fpage>11441</fpage>. <pub-id pub-id-type="doi">10.1073/pnas.1604850113</pub-id><pub-id pub-id-type="pmid">27651489</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Mao</surname> <given-names>H.</given-names></name> <name><surname>Pu</surname> <given-names>J.</given-names></name> <name><surname>Pedram</surname> <given-names>A.</given-names></name> <name><surname>Horowitz</surname> <given-names>M. A.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Eie: Efficient inference engine on compressed deep neural network</article-title>. <source>ACM SIGARCH Comp. Architect. News</source> <volume>44</volume>, <fpage>243</fpage>&#x02013;<lpage>254</lpage>. <pub-id pub-id-type="doi">10.1145/3007787.3001163</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Howard</surname> <given-names>A. G.</given-names></name> <name><surname>Zhu</surname> <given-names>M.</given-names></name> <name><surname>Chen</surname> <given-names>B.</given-names></name> <name><surname>Kalenichenko</surname> <given-names>D.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Weyand</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Mobilenets: efficient convolutional neural networks for mobile vision applications</article-title>. <source>arXiv</source> [preprint] arXiv:1704.04861.</citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>X.</given-names></name> <name><surname>Tang</surname> <given-names>K.</given-names></name> <name><surname>Miao</surname> <given-names>C.</given-names></name> <name><surname>Hua</surname> <given-names>X.-S.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Distilling causal effect of data in class-incremental learning,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Nashville, TN</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>3957</fpage>&#x02013;<lpage>3966</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>C.-C.</given-names></name> <name><surname>Jin</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Swapadvisor: Pushing deep learning beyond the gpu memory limit via smart swapping,&#x0201D;</article-title> in <source>Proceedings of the Twenty-Fifth International Conference on Architectural Support for Programming Languages and Operating Systems, ASPLOS &#x00027;20</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>1341</fpage>&#x02013;<lpage>1355</lpage>. <pub-id pub-id-type="doi">10.1145/3373376.3378530</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>W.</given-names></name> <name><surname>Ma</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Zhou</surname> <given-names>B. B.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Layup: Layer-adaptive and multi-type intermediate-oriented memory optimization for gpu-based cnns</article-title>. <source>ACM Trans. Architect. Code Optimizat</source>. (<italic>TACO)</italic> 16(4):1&#x02013;23. <pub-id pub-id-type="doi">10.1145/3357238</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jouppi</surname> <given-names>N. P.</given-names></name> <name><surname>Yoon</surname> <given-names>D. H.</given-names></name> <name><surname>Ashcraft</surname> <given-names>M.</given-names></name> <name><surname>Gottscho</surname> <given-names>M.</given-names></name> <name><surname>Jablin</surname> <given-names>T. B.</given-names></name> <name><surname>Kurian</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Ten lessons from three generations shaped google&#x00027;s tpuv4i: Industrial product,&#x0201D;</article-title> in <source>2021 ACM/IEEE 48th Annual International Symposium on Computer Architecture (ISCA)</source> (<publisher-loc>Valencia</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>14</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jouppi</surname> <given-names>N. P.</given-names></name> <name><surname>Young</surname> <given-names>C.</given-names></name> <name><surname>Patil</surname> <given-names>N.</given-names></name> <name><surname>Patterson</surname> <given-names>D.</given-names></name> <name><surname>Agrawal</surname> <given-names>G.</given-names></name> <name><surname>Bajwa</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>&#x0201C;In-datacenter performance analysis of a tensor processing unit,&#x0201D;</article-title> in <source>Proceedings of the 44th Annual International Symposium on Computer Architecture</source> (<publisher-loc>Toronto, ON</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>12</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kung</surname> <given-names>H.</given-names></name> <name><surname>McDanel</surname> <given-names>B.</given-names></name> <name><surname>Zhang</surname> <given-names>S. Q.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Packing sparse convolutional neural networks for efficient systolic array implementations: column combining under joint optimization,&#x0201D;</article-title> in <source>Proceedings of the Twenty-Fourth International Conference on Architectural Support for Programming Languages and Operating Systems, ASPLOS &#x00027;19</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>821</fpage>&#x02013;<lpage>834</lpage>. <pub-id pub-id-type="doi">10.1145/3297858.3304028</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lakshmi</surname> <given-names>V.</given-names></name> <name><surname>Reuben</surname> <given-names>J.</given-names></name> <name><surname>Pudi</surname> <given-names>V.</given-names></name></person-group> (<year>2021</year>). <article-title>A novel in-memory wallace tree multiplier architecture using majority logic</article-title>. <source>IEEE Trans. Circuits Syst. I: Regular Papers</source> <volume>69</volume>, <fpage>1148</fpage>&#x02013;<lpage>1158</lpage>. <pub-id pub-id-type="doi">10.1109/TCSI.2021.3129827</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Le Gallo</surname> <given-names>M.</given-names></name> <name><surname>Khaddam-Aljameh</surname> <given-names>R.</given-names></name> <name><surname>Stanisavljevic</surname> <given-names>M.</given-names></name> <name><surname>Vasilopoulos</surname> <given-names>A.</given-names></name> <name><surname>Kersting</surname> <given-names>B.</given-names></name> <name><surname>Dazzi</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>A 64-core mixed-signal in-memory compute chip based on phase-change memory for deep neural network inference</article-title>. <source>Nat. Electron</source>. <volume>2023</volume>, <fpage>1</fpage>&#x02013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1038/s41928-023-01010-1</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>C.-K.</given-names></name> <name><surname>Wild</surname> <given-names>A.</given-names></name> <name><surname>Chinya</surname> <given-names>G. N.</given-names></name> <name><surname>Lin</surname> <given-names>T.-H.</given-names></name> <name><surname>Davies</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Mapping spiking neural networks onto a manycore neuromorphic architecture</article-title>. <source>ACM SIGPLAN Notices</source> <volume>53</volume>, <fpage>78</fpage>&#x02013;<lpage>89</lpage>. <pub-id pub-id-type="doi">10.1145/3296979.3192371</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Schiele</surname> <given-names>B.</given-names></name> <name><surname>Sun</surname> <given-names>Q.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;RMM: reinforced memory management for class-incremental learning,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems, vol. 34</source>, eds. M. Ranzato, A. Beygelzimer, Y. Dauphin, P. Liang, and J. W. Vaughan (Curran Associates, Inc.), <fpage>3478</fpage>&#x02013;<lpage>3490</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>S.</given-names></name> <name><surname>Pei</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>G.</given-names></name> <name><surname>Feng</surname> <given-names>D.</given-names></name> <name><surname>Yu</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Neuromorphic computing chip with spatiotemporal elasticity for multi-intelligent-tasking robots</article-title>. <source>Sci. Robot</source>. <volume>7</volume>:<fpage>eabk2948</fpage>. <pub-id pub-id-type="doi">10.1126/scirobotics.abk2948</pub-id><pub-id pub-id-type="pmid">35704609</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Modha</surname> <given-names>D. S.</given-names></name> <name><surname>Akopyan</surname> <given-names>F.</given-names></name> <name><surname>Andreopoulos</surname> <given-names>A.</given-names></name> <name><surname>Appuswamy</surname> <given-names>R.</given-names></name> <name><surname>Arthur</surname> <given-names>J. V.</given-names></name> <name><surname>Cassidy</surname> <given-names>A. S.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Neural inference at the frontier of energy, space, and time</article-title>. <source>Science</source> <volume>382</volume>, <fpage>329</fpage>&#x02013;<lpage>335</lpage>. <pub-id pub-id-type="doi">10.1126/science.adh1174</pub-id><pub-id pub-id-type="pmid">37856600</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Myung</surname> <given-names>W.</given-names></name> <name><surname>Lee</surname> <given-names>D.</given-names></name> <name><surname>Song</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>G.</given-names></name> <name><surname>Ma</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>Policy gradient-based core placement optimization for multichip many-core systems</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst</source>. <volume>34</volume>, <fpage>4529</fpage>&#x02013;<lpage>4543</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2021.3117878</pub-id><pub-id pub-id-type="pmid">34644256</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Nie</surname> <given-names>X.</given-names></name> <name><surname>Miao</surname> <given-names>X.</given-names></name> <name><surname>Yang</surname> <given-names>Z.</given-names></name> <name><surname>Cui</surname> <given-names>B.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;TSPLIT: Fine-grained gpu memory management for efficient dnn training via tensor splitting,&#x0201D;</article-title> in <source>2022 IEEE 38th International Conference on Data Engineering (ICDE)</source> (<publisher-loc>Kuala Lumpur</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2615</fpage>&#x02013;<lpage>2628</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pei</surname> <given-names>J.</given-names></name> <name><surname>Deng</surname> <given-names>L.</given-names></name> <name><surname>Song</surname> <given-names>S.</given-names></name> <name><surname>Zhao</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Towards artificial general intelligence with hybrid tianjic chip architecture</article-title>. <source>Nature</source> <volume>572</volume>, <fpage>106</fpage>&#x02013;<lpage>111</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-019-1424-8</pub-id><pub-id pub-id-type="pmid">31367028</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rhu</surname> <given-names>M.</given-names></name> <name><surname>Gimelshein</surname> <given-names>N.</given-names></name> <name><surname>Clemons</surname> <given-names>J.</given-names></name> <name><surname>Zulfiqar</surname> <given-names>A.</given-names></name> <name><surname>Keckler</surname> <given-names>S. W.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;vDNN: Virtualized deep neural networks for scalable, memory-efficient neural network design,&#x0201D;</article-title> in <source>2016 49th Annual IEEE/ACM International Symposium on Microarchitecture (MICRO)</source> (<publisher-loc>Taipei</publisher-loc>,: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>13</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rueckauer</surname> <given-names>B.</given-names></name> <name><surname>Bybee</surname> <given-names>C.</given-names></name> <name><surname>Goettsche</surname> <given-names>R.</given-names></name> <name><surname>Singh</surname> <given-names>Y.</given-names></name> <name><surname>Mishra</surname> <given-names>J.</given-names></name> <name><surname>Wild</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>NxTF: an API and compiler for deep spiking neural networks on intel loihi</article-title>. <source>ACM J. Emerg. Technol. Comp. Syst. (JETC)</source> <volume>18</volume>, <fpage>1</fpage>&#x02013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1145/3501770</pub-id><pub-id pub-id-type="pmid">33162886</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sawada</surname> <given-names>J.</given-names></name> <name><surname>Akopyan</surname> <given-names>F.</given-names></name> <name><surname>Cassidy</surname> <given-names>A. S.</given-names></name> <name><surname>Taba</surname> <given-names>B.</given-names></name> <name><surname>Debole</surname> <given-names>M. V.</given-names></name> <name><surname>Datta</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>&#x0201C;TrueNorth ecosystem for brain-inspired computing: scalable systems, software, and applications,&#x0201D;</article-title> in <source>SC&#x00027;16: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis</source> (<publisher-loc>Salt Lake City</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>130</fpage>&#x02013;<lpage>141</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Shao</surname> <given-names>Y. S.</given-names></name> <name><surname>Clemons</surname> <given-names>J.</given-names></name> <name><surname>Venkatesan</surname> <given-names>R.</given-names></name> <name><surname>Zimmer</surname> <given-names>B.</given-names></name> <name><surname>Fojtik</surname> <given-names>M.</given-names></name> <name><surname>Jiang</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Simba: Scaling deep-learning inference with multi-chip-module-based architecture,&#x0201D;</article-title> in <source>Proceedings of the 52nd Annual IEEE/ACM International Symposium on Microarchitecture</source> (<publisher-loc>Columbus, Ohio</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>14</fpage>&#x02013;<lpage>27</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>J.</given-names></name> <name><surname>Ma</surname> <given-names>D.</given-names></name> <name><surname>Gu</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Zhu</surname> <given-names>X.</given-names></name> <name><surname>Xu</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Darwin: a neuromorphic hardware co-processor based on spiking neural networks</article-title>. <source>Science China Information Sciences</source> 59(2):1&#x02013;5. <pub-id pub-id-type="doi">10.1007/s11432-015-5511-7</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Shrestha</surname> <given-names>S. B.</given-names></name> <name><surname>Timcheck</surname> <given-names>J.</given-names></name> <name><surname>Frady</surname> <given-names>P.</given-names></name> <name><surname>Campos-Macias</surname> <given-names>L.</given-names></name> <name><surname>Davies</surname> <given-names>M.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;Efficient video and audio processing with Loihi 2,&#x0201D;</article-title> in <source>ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</source> (<publisher-loc>Seoul</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>13481</fpage>&#x02013;<lpage>13485</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Solanki</surname> <given-names>V.</given-names></name> <name><surname>Darji</surname> <given-names>A. D.</given-names></name> <name><surname>Singapuri</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>Design of low-power wallace tree multiplier architecture using modular approach</article-title>. <source>Circu. Syst. Signal Proc</source>. <volume>40</volume>, <fpage>4407</fpage>&#x02013;<lpage>4427</lpage>. <pub-id pub-id-type="doi">10.1007/s00034-021-01671-3</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Srinivas</surname> <given-names>L.</given-names></name> <name><surname>Umapathi</surname> <given-names>N.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;New realization of low area and high-performance wallace tree multipliers using booth recoding unit,&#x0201D;</article-title> in <source>AIP Conference Proceedings</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>AIP Publishing</publisher-name>).</citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sze</surname> <given-names>V.</given-names></name> <name><surname>Chen</surname> <given-names>Y.-H.</given-names></name> <name><surname>Yang</surname> <given-names>T.-J.</given-names></name> <name><surname>Emer</surname> <given-names>J. S.</given-names></name></person-group> (<year>2017</year>). <article-title>Efficient processing of deep neural networks: a tutorial and survey</article-title>. <source>Proc. IEEE</source> <volume>105</volume>, <fpage>2295</fpage>&#x02013;<lpage>2329</lpage>. <pub-id pub-id-type="doi">10.1109/JPROC.2017.2761740</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wahib</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Nguyen</surname> <given-names>T. T.</given-names></name> <name><surname>Drozd</surname> <given-names>A.</given-names></name> <name><surname>Domke</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>&#x0201C;Scaling distributed deep learning workloads beyond the memory capacity with karma,&#x0201D;</article-title> in <source>SC20: International Conference for High Performance Computing, Networking, Storage and Analysis</source> (<publisher-loc>Atlanta, GA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>15</lpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>G.</given-names></name> <name><surname>Ma</surname> <given-names>S.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Pei</surname> <given-names>J.</given-names></name> <name><surname>Zhao</surname> <given-names>R.</given-names></name> <name><surname>Shi</surname> <given-names>L.</given-names></name></person-group> (<year>2021</year>). <article-title>End-to-end implementation of various hybrid neural networks on a cross-paradigm neuromorphic chip</article-title>. <source>Front. Neurosci</source>. <volume>15</volume>:<fpage>615279</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2021.615279</pub-id><pub-id pub-id-type="pmid">33603643</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Kan</surname> <given-names>M.</given-names></name> <name><surname>Shan</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Fully learnable group convolution for acceleration of deep neural networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Matilda</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>9049</fpage>&#x02013;<lpage>9058</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Lu</surname> <given-names>J.</given-names></name> <name><surname>Jiang</surname> <given-names>H.</given-names></name> <name><surname>An</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Fangtiansim: High-level cycle-accurate resistive random-access memory-based multi-core spiking neural network processor simulator</article-title>. <source>Front. Neurosci</source>. <volume>15</volume>:<fpage>806325</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2021.806325</pub-id><pub-id pub-id-type="pmid">35126046</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>N.</given-names></name> <name><surname>Deng</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>G.</given-names></name> <name><surname>Xie</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Core placement optimization for multi-chip many-core neural network systems with reinforcement learning</article-title>. <source>ACM Trans. Design Autom. Elect. Syst</source>. <volume>26</volume>, <fpage>1</fpage>&#x02013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1145/3418498</pub-id></citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Shi</surname> <given-names>B.</given-names></name> <name><surname>Zheng</surname> <given-names>Z.</given-names></name> <name><surname>Zheng</surname> <given-names>H.</given-names></name> <name><surname>Yu</surname> <given-names>F.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Adaptive spatiotemporal neural networks through complementary hybridization</article-title>. <source>Nat. Commun</source>. <volume>15</volume>:<fpage>7355</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-024-51641-x</pub-id><pub-id pub-id-type="pmid">39191782</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Xiao</surname> <given-names>W.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Hou</surname> <given-names>P.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2020</year>). <source>Antman: Dynamic Scaling on GPU Clusters for Deep Learning</source>. <publisher-loc>Boston, MA</publisher-loc>: <publisher-name>OSDI</publisher-name>, <fpage>533</fpage>&#x02013;<lpage>548</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Xie</surname> <given-names>S.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name> <name><surname>Tu</surname> <given-names>Z.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Aggregated residual transformations for deep neural networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1492</fpage>&#x02013;<lpage>1500</lpage>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>S.</given-names></name> <name><surname>Ouyang</surname> <given-names>P.</given-names></name> <name><surname>Tang</surname> <given-names>S.</given-names></name> <name><surname>Tu</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Zheng</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>A high energy efficient reconfigurable hybrid neural network processor for deep learning applications</article-title>. <source>IEEE J. Solid-State Circuits</source> <volume>53</volume>, <fpage>968</fpage>&#x02013;<lpage>982</lpage>. <pub-id pub-id-type="doi">10.1109/JSSC.2017.2778281</pub-id><pub-id pub-id-type="pmid">27534393</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>F.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Ma</surname> <given-names>S.</given-names></name> <name><surname>Xu</surname> <given-names>M.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Qu</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Brain-inspired multimodal hybrid neural network for robot place recognition</article-title>. <source>Sci. Robot</source>. <volume>8</volume>:<fpage>eabm6996</fpage>. <pub-id pub-id-type="doi">10.1126/scirobotics.abm6996</pub-id><pub-id pub-id-type="pmid">37163608</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Shao</surname> <given-names>W.</given-names></name> <name><surname>Peng</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>R.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Differentiable learning-to-group channels via groupable convolutional neural networks,&#x0201D;</article-title> in <source>2019 IEEE/CVF International Conference on Computer Vision (ICCV)</source> (<publisher-loc>Los Alamitos, CA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>), <fpage>3541</fpage>&#x02013;<lpage>3550</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2019.00364</pub-id></citation>
</ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>H.</given-names></name> <name><surname>Zheng</surname> <given-names>Z.</given-names></name> <name><surname>Hu</surname> <given-names>R.</given-names></name> <name><surname>Xiao</surname> <given-names>B.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Yu</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Temporal dendritic heterogeneity incorporated with spiking neural networks for learning multi-timescale dynamics</article-title>. <source>Nat. Commun</source>. <volume>15</volume>:<fpage>277</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-023-44614-z</pub-id><pub-id pub-id-type="pmid">38177124</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhong</surname> <given-names>Y.</given-names></name> <name><surname>Kuang</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>K.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Feng</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Paicore: a 1.9-million-neuron 5.181-tsops/w digital neuromorphic processor with unified snn-ann and on-chip learning paradigm</article-title>. <source>IEEE J. Solid-State Circ</source>. <volume>60</volume>, <fpage>651</fpage>&#x02013;<lpage>671</lpage>. <pub-id pub-id-type="doi">10.1109/JSSC.2024.3426319</pub-id></citation>
</ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zimmer</surname> <given-names>B.</given-names></name> <name><surname>Venkatesan</surname> <given-names>R.</given-names></name> <name><surname>Shao</surname> <given-names>Y. S.</given-names></name> <name><surname>Clemons</surname> <given-names>J.</given-names></name> <name><surname>Fojtik</surname> <given-names>M.</given-names></name> <name><surname>Jiang</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>A 0.32-128 tops, scalable multi-chip-module-based deep neural network inference accelerator with ground-referenced signaling in 16 nm</article-title>. <source>IEEE J. Solid-State Circuits</source> <volume>55</volume>, <fpage>920</fpage>&#x02013;<lpage>932</lpage>. <pub-id pub-id-type="doi">10.1109/JSSC.2019.2960488</pub-id></citation>
</ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>C.</given-names></name> <name><surname>Cui</surname> <given-names>X.</given-names></name> <name><surname>Kuang</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>K.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>A scatter-and-gather spiking convolutional neural network on a reconfigurable neuromorphic hardware</article-title>. <source>Front. Neurosci</source>. <volume>15</volume>:<fpage>694170</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2021.694170</pub-id><pub-id pub-id-type="pmid">34867142</pub-id></citation></ref>
</ref-list>
</back>
</article>