<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurorobot.</journal-id>
<journal-title>Frontiers in Neurorobotics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurorobot.</abbrev-journal-title>
<issn pub-type="epub">1662-5218</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnbot.2025.1537673</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>High-efficiency sparse convolution operator for event-based cameras</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Sen</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2910269/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zha</surname> <given-names>Fusheng</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/799501/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Xiangji</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2912814/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Li</surname> <given-names>Mantian</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Guo</surname> <given-names>Wei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Pengfei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Xiaolin</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Sun</surname> <given-names>Lining</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>State Key Laboratory of Robotics and System, Harbin Institute of Technology</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Lanzhou University of Technology</institution>, <addr-line>Lanzhou</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Institute of Intelligent Manufacturing Technology, Shenzhen Polytechnic University</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Alois C. Knoll, Technical University of Munich, Germany</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Bo You, Harbin University of Science and Technology, China</p>
<p>Wenzheng Chi, Soochow University, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Fusheng Zha <email>zhafusheng&#x00040;hit.edu.cn</email></corresp>
<corresp id="c002">Mantian Li <email>limtsz&#x00040;szpu.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>12</day>
<month>03</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>19</volume>
<elocation-id>1537673</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>12</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>02</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Zhang, Zha, Wang, Li, Guo, Wang, Li and Sun.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zhang, Zha, Wang, Li, Guo, Wang, Li and Sun</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Event-based cameras are bio-inspired vision sensors that mimic the sparse and asynchronous activation of the animal retina, offering advantages such as low latency and low computational load in various robotic applications. However, despite their inherent sparsity, most existing visual processing algorithms are optimized for conventional standard cameras and dense images captured from them, resulting in computational redundancy and high latency when applied to event-based cameras. To address this gap, we propose a sparse convolution operator tailored for event-based cameras. By selectively skipping invalid sub-convolutions and efficiently reorganizing valid computations, our operator reduces computational workload by nearly 90% and achieves almost 2&#x000D7; acceleration in processing speed, while maintaining the same accuracy as dense convolution operators. This innovation unlocks the potential of event-based cameras in applications such as autonomous navigation, real-time object tracking, and industrial inspection, enabling low-latency and high-efficiency perception in resource-constrained robotic systems.</p></abstract>
<kwd-group>
<kwd>event-based camera</kwd>
<kwd>sparse convolution</kwd>
<kwd>convolution operator</kwd>
<kwd>high-efficiency</kwd>
<kwd>low-latency</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="0"/>
<equation-count count="7"/>
<ref-count count="29"/>
<page-count count="10"/>
<word-count count="7177"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Low-computation and low-latency visual perception are crucial for robotic systems. Compared to the highly efficient visual processing capabilities of advanced biological systems such as humans, robotic vision often requires significantly greater computational resources (Wu et al., <xref ref-type="bibr" rid="B26">2022</xref>; Meng et al., <xref ref-type="bibr" rid="B13">2025</xref>). This inefficiency imposes substantial constraints on a wide range of robotic applications. A notable example can be seen in small micro aerial vehicles (MAVs). To maximize flight endurance, MAVs are typically designed to be lightweight, which restricts them to carrying low-power embedded computing devices. As a result, their visual processing capabilities are often limited to basic visual functions or extremely slow in high-level 3D vision (Guo et al., <xref ref-type="bibr" rid="B9">2024</xref>; Cheng et al., <xref ref-type="bibr" rid="B4">2024</xref>), frequently making timely and effective obstacle avoidance difficult. The challenge becomes even more pronounced in autonomous driving. Ensuring safety and robustness demands the simultaneous execution of multiple perception sub-tasks (Qian et al., <xref ref-type="bibr" rid="B21">2022</xref>; Wang et al., <xref ref-type="bibr" rid="B25">2019</xref>; Jiang et al., <xref ref-type="bibr" rid="B11">2019</xref>). Moreover, integrating information across spatial, color, temporal, and multi-camera dimensions significantly increases the computational burden, making real-time perception increasingly difficult. Consequently, autonomous driving systems often struggle to meet stringent latency requirements. Their typical perception update rates reach only around 30 Hz (Yu et al., <xref ref-type="bibr" rid="B28">2023</xref>), which falls far short of the ideal requirements.</p>
<p>An event-based camera is a bio-inspired vision sensor designed based on the working principles of the animal retina. Drawing inspiration from the transient visual pathway, it replicates the sparse activation and asynchronous transmission of retinal ganglion cells. Unlike conventional cameras that capture visual information by recording entire image frames, event-based cameras encode visual data through discrete &#x0201C;event&#x0201D;, as illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>. This approach offers a finer and more dynamic representation of visual information, akin to how visual neurons in the animal nervous system respond selectively to changes in their environment. To achieve this, the pixels of an event-based camera remain continuously exposed, much like the human eye&#x00027;s constant reception of light. When the perceived light intensity fluctuates beyond a certain threshold, the corresponding pixel generates a signal. Depending on whether the intensity increases or decreases, this signal can be positive or negative. These signals, known as &#x0201C;event&#x0201D;, serve as the fundamental units of visual information in an event-based camera. An &#x0201C;event&#x0201D; captures and encodes significant changes in light intensity from the environment, selectively highlighting the most essential visual information. This biologically inspired strategy results in a highly sparse output, with visual information at any moment far less than that of conventional cameras (Rebecq et al., <xref ref-type="bibr" rid="B22">2019</xref>; Miao et al., <xref ref-type="bibr" rid="B15">2019</xref>). As shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, sparsity often reaches 99%, enabling efficient processing by reducing computational complexity and inference time.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Working principle of event-based cameras. <bold>(A)</bold> demonstrates working principle of a single pixel, while <bold>(B)</bold> shows the comparison of imaging results between event-based cameras and conventional standard cameras.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1537673-g0001.tif"/>
</fig>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Spatial sparsity of event-based cameras, <bold>(A)</bold> demonstrates the events in spatial-temporal coordinate. <bold>(B&#x02013;D)</bold> shows the 2D histogram image by accumulating events in time windows of 1 ms, 10 ms, and 100 ms. They shows the great sparsity comparing with dense image captured by conventional standard cameras.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1537673-g0002.tif"/>
</fig>
<p>However, despite event-based cameras have the characters of low latency and high sparsity, vision processing algorithms that fully leverage this sparsity are still relatively rare. Current robotic vision algorithms are primarily designed for conventional dense vision systems. Since event-based cameras operate on fundamentally different principles than standard cameras, applying traditional vision algorithms to event-based camera data often disregards their inherent sparsity. This leads to high computational complexity and increased latency, resulting in inefficient use of computational resources and longer inference times, ultimately hindering the potential benefits of sparse, low-latency processing. Therefore, a critical challenge lies in how to effectively utilize the sparse nature of event-based camera data and design sparse processing algorithms specifically tailored for event-based data.</p>
<p>Some research efforts have focused on leveraging the sparsity of event-based cameras and reducing redundant computations. One approach processes events sequentially in real-time (Brosch et al., <xref ref-type="bibr" rid="B3">2015</xref>), eliminating zero-padding and improving efficiency, but requiring extensive manual design, limiting generalization to complex tasks. CNNs remain dominant in visual perception, with studies showing improved accuracy for event-based data. To optimize convolutions, some methods use hash tables (Messikommer et al., <xref ref-type="bibr" rid="B14">2020</xref>) to manage sparse computations efficiently. However, hash lookups disrupt computational continuity (Sorin et al., <xref ref-type="bibr" rid="B24">2022</xref>), leading to increased inference latency despite reduced computation. Similarly, Graph Convolutional Networks (GCNs) (Schaefer et al., <xref ref-type="bibr" rid="B23">2022</xref>) reduce computational load but also suffer from discontinuous processing, preventing latency improvements. Spiking Neural Networks (SNNs) (Cordone et al., <xref ref-type="bibr" rid="B5">2021</xref>; Orchard et al., <xref ref-type="bibr" rid="B18">2015b</xref>; Bing et al., <xref ref-type="bibr" rid="B2">2018</xref>; Jiang et al., <xref ref-type="bibr" rid="B10">2017</xref>), inspired by biological neurons, align well with event-based data due to sparse activation and asynchronous updates. However, they struggle with accuracy compared to CNNs and often require specialized hardware, limiting their practical use. Despite progress, no existing method fully resolves the balance among computational efficiency, low latency, and high accuracy in event-based vision.</p>
<p>To achieve the goal of low-latency visual perception with minimal computational resources while maintaining high accuracy, this paper proposes a sparse convolution operator specifically designed for event-based cameras. By eliminating redundant sub-convolutions, the computational load is significantly reduced. Meanwhile, the valid sub-convolutions are efficiently reorganized into matrix multiplication operations, greatly enhancing the computation speed while preserving the same level of accuracy as conventional convolution operators.</p>
<p>The main contributions of this paper are as follows: 1. Sparse Convolution Operator: We propose a novel sparse convolution operator that significantly reduces the computational load and accelerates inference for sparse input data. To the best of our knowledge, this is the first sparse convolution operator that surpasses the inference speed of conventional convolution operators when processing event-based camera data. 2. Efficient Sub-Convolution Detection: We introduce an efficient method for detecting valid sub-convolutions based on the location information of active pixels, enabling the computation of sub-convolution indices without the need for exhaustive traversal. 3. Sparse im2col: We present a sparse im2col technique that reorganizes sparse convolution input data into a dense matrix format, allowing for efficient matrix multiplication, protecting the computational continuity and further accelerating the sparse convolution operation.</p></sec>
<sec id="s2">
<title>2 Related works</title>
<sec>
<title>2.1 Sparse convolution</title>
<p>Image-based visual data often encounters sparse input situations, such as handwritten digits, 3D point clouds, and 3D voxel data. These types of data contain large areas of empty space, with meaningful information concentrated in only a small portion of the region. When using conventional convolutional neural networks to process this data, a large amount of empty, invalid computations are generated. Therefore, a class of methods attempts to modify the convolution operator by discarding invalid operations to take advantage of this data sparsity, aiming to reduce computation and accelerate processing. These methods are called sparse convolutions. The most classic sparse convolution approach was first proposed by Facebook (Graham and Van der Maaten, <xref ref-type="bibr" rid="B8">2017</xref>; Graham et al., <xref ref-type="bibr" rid="B7">2018</xref>). The authors, when solving the handwritten digit recognition problem, noticed the sparsity of the data and used hash tables to record the elements to be computed and their computational relationships, reflecting the sparsity. Since the table only records valid computations and excludes invalid operations in the empty regions, this method successfully reduces the computational load. This approach was then applied to the processing of 3D point clouds and 3D voxels (Yan et al., <xref ref-type="bibr" rid="B27">2018</xref>). In practical scenarios, 3D point cloud information is generally highly sparse, with large areas of empty space. By extending the above method to three dimensions, sparse convolution has been successfully applied in the field of autonomous driving for 3D point cloud recognition, greatly improving the efficiency of 3D convolution operations for point clouds and voxels. It is worth noting that the creation of the hash table itself also consumes computational resources, and the advantages of this method are only evident when the input data is sufficiently large. For 3D point clouds, the computational delay caused by the dimensional explosion of 3D convolution is very high, highlighting the benefits of sparsity. However, for ordinary 2D images, the creation cost of the hash table is generally non-negligible. Therefore, for event-based cameras, sparse convolution based on hash tables is difficult to fully realize its potential in terms of computational delay. In addition to the hash table-based sparse convolution approach, there are other sparse convolution applications. For example, Parger et al. (<xref ref-type="bibr" rid="B20">2022b</xref>,<xref ref-type="bibr" rid="B19">a</xref>) noticed that similar content appears frequently between consecutive video frames, so the difference between two frames is used as the input data for each inference. This difference often exhibits spatial proximity and is likely to satisfy computational continuity, where sparse convolution can achieve good results with a simple approach. However, the effective data generated by event-based cameras is more unevenly distributed, making it less suitable for simple sparse convolution methods.</p></sec>
<sec>
<title>2.2 Data processing for event-based cameras</title>
<p>Due to the significant differences in the working principles and information representation between event-based cameras and standard cameras, corresponding visual perception algorithms also exhibit significant differences. Generally speaking, event-based camera data processing algorithms can be divided into two categories: one is event-based processing, and the other is to aggregate events into groups for processing (Gallego et al., <xref ref-type="bibr" rid="B6">2020</xref>). In general, event-based processing better leverages the high event resolution and low latency advantages of event-based cameras, as there is little waiting time between event generation and processing. One typical application of this approach is in SLAM systems. By utilizing methods like Kalman filtering or particle filtering, robot pose tracking can be efficiently and quickly performed from event data. Event-based processing can also be applied to other vision tasks, such as feature extraction (Brosch et al., <xref ref-type="bibr" rid="B3">2015</xref>) and image reconstruction (Munda et al., <xref ref-type="bibr" rid="B16">2018</xref>). These tasks combine past event information with current data to accomplish high-level visual tasks, which to some extent align with the asynchronous nature of event-based cameras. However, a drawback of event-based processing is that it cannot provide enough effective information at once and is susceptible to noise interference, often limiting its application range and performance, particularly in high-level semantic reasoning. Aggregating events into groups allows for the simultaneous consideration of more information, making the information extraction and reasoning process more convenient. The methods of event aggregation and representation are diverse, mainly including event frames (Liu and Delbruck, <xref ref-type="bibr" rid="B12">2018</xref>), Time Surface, Voxel Grid, 3D Point Sets, and others. The event frame representation allows for the reuse of conventional image processing methods, such as convolutional neural networks (CNNs), to process event-based camera data, and such methods have been proven to be highly effective in various tasks. Time Surface is a representation sensitive to motion direction and scene edges, making it particularly effective in applications such as optical flow estimation (Benosman et al., <xref ref-type="bibr" rid="B1">2013</xref>). Many studies now use this approach as the foundation for dynamic feature extraction, feeding it into CNNs and other neural networks for perceptual reasoning, achieving promising results (Zhu et al., <xref ref-type="bibr" rid="B29">2018</xref>). Voxel Grid and 3D Point Set representations extend the information into the spatiotemporal domain. On one hand, they retain more information; on the other hand, they demand higher computational resources. These representations can generally be input into various ANN models (e.g., CNNs) for better processing results. From this, it can be seen that convolutional neural networks are foundational modules in event-based camera data processing methods, and convolution operators play an important role in event-based camera perception and reasoning. Moreover, there are still relatively few approaches that focus on the sparsity of event data in convolutional inference.</p></sec></sec>
<sec sec-type="methods" id="s3">
<title>3 Methods</title>
<p>This paper is primarily inspired by the GEMM (General Matrix Multiply) method in traditional dense convolutions. By transforming the sparse convolution for processing event-based camera data into dense matrix multiplication, it reduces the computational load while preserving the continuity of the computations, thus achieving efficient sparse convolution operations. The following section provides a detailed explanation of the specific approach.</p>
<sec>
<title>3.1 Converting events to tensor</title>
<p>The main difference in working principles between event-based cameras and standard cameras lies in the fact that in an event-based camera, each pixel unit operates independently and asynchronously. The data generation process of each pixel unit is not controlled by a unified clock cycle, but rather by the changes in the ambient light information. The basic perceptual output of each pixel unit is called an event. An event is triggered when the logarithmic intensity of the light stimulus received by a pixel unit exceeds a preset threshold <italic>c</italic> compared to the previous moment, that is:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>|</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:mo>&#x0003E;</mml:mo><mml:mi>c</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>It will then generate and output <italic>event</italic><sub><italic>i</italic></sub>, which contains the pixel coordinates (<italic>x</italic><sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub>) of the pixel, the time <italic>t</italic><sub><italic>i</italic></sub> when the event occurred, and the polarity <italic>p</italic><sub><italic>i</italic></sub>, which indicates whether the change in light intensity exceeded the threshold in an upward or downward direction. That is:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>e</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>As the ambient light information changes, the event-based camera can output a continuous asynchronous event stream over time. A single event contains very little information, making it difficult to achieve the desired results through processing individual events. However, by recording all events that occur within a small time window &#x00394;<italic>t</italic>, a set of events, <italic>S</italic><sub>&#x00394;<italic>t</italic></sub>, can be obtained. This event set <italic>S</italic><sub>&#x00394;<italic>t</italic></sub> carries more information and provides a more meaningful representation for further processing.</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mtext>&#x00394;</mml:mtext><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">{</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mtext>&#x00394;</mml:mtext><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>When the ambient light source remains constant, the changes in brightness within an image are typically caused by the movement of objects. Specifically, movement at the edges of objects tends to result in more significant changes in intensity. Therefore, the event set <italic>S</italic> generally contains information about object edges, which is crucial for visual tasks such as object detection. By using a sliding window, the contents of <italic>S</italic> can be continuously updated to acquire new perceptual data. Additionally, by controlling &#x00394;<italic>t</italic>, one can adjust the temporal receptive field and the amount of spatiotemporal information processed in each instance.</p>
<p>Modern AI algorithms typically use tensors as the fundamental data structure. Considering the scalability and compatibility of the proposed convolution operator, event-based camera data <italic>S</italic> is expressed here in the form of an image tensor, similar to traditional images. Based on the definitions above, each <italic>event</italic><sub><italic>i</italic></sub> contains information in four dimensions. First, for the time <italic>t</italic>-dimension, when &#x00394;<italic>t</italic> is sufficiently small, it can be approximated that the events in <italic>S</italic> occur simultaneously. In this case, the time-axis information of visual events within the &#x00394;<italic>t</italic> time window is compressed. For the polarity dimension, the polarity <italic>p</italic> of an event has only two possible values, and events with the same polarity <italic>p</italic> are spatially and logically closely related. Therefore, events with different polarities <italic>p</italic> can be treated as information in different channels of the image, i.e., using polarity as the channel dimension. Finally, for the width and height dimensions, the <italic>x</italic> and <italic>y</italic> coordinates of <italic>event</italic><sub><italic>i</italic></sub> represent the pixel positions where the event occurs, which is consistent with traditional images. Thus, an event <italic>event</italic><sub><italic>i</italic></sub> can be represented as a multi-dimensional vector <italic>T</italic><sub><italic>i</italic></sub>, similar to a pixel in a traditional image.</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>T</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In this representation, <italic>p</italic><sub><italic>i</italic></sub>, <italic>h</italic><sub><italic>i</italic></sub>, and <italic>w</italic><sub><italic>i</italic></sub> belong to event<sub><italic>i</italic></sub>, where <italic>N</italic> represents the batch dimension, <italic>p</italic><sub><italic>i</italic></sub> is the channel dimension, and <italic>h</italic><sub><italic>i</italic></sub> and <italic>w</italic><sub><italic>i</italic></sub> represent the height and width dimensions, respectively. The tensor <italic>T</italic><sub><italic>i</italic></sub> obtained from all events that occurred within the time window &#x00394;<italic>t</italic> can be stacked to form the real-valued part of the image tensor. However, there are many pixel positions within the time window &#x00394;<italic>t</italic> where no visual events occurred. For these positions where no event has taken place, zeros are used to fill the corresponding locations. In this way, we obtain a data representation in the form of a tensor <italic>ET</italic>, which is compatible with modern computer vision techniques. This tensor <italic>ET</italic> provides a structured representation of event-based camera data that can be processed using standard AI algorithms.</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>E</mml:mi><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mo>&#x02211;</mml:mo><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The <italic>x</italic><sub><italic>i</italic></sub> and <italic>y</italic><sub><italic>i</italic></sub> in <italic>event</italic><sub><italic>i</italic></sub> record the positions of valid pixels in <italic>ET</italic>. We will denote the set of all valid pixel positions in <italic>ET</italic> as <italic>Valid</italic>_<italic>pix</italic>. These are the positions where events have occurred and contain meaningful visual information, distinguishing them from the positions that remain empty (where no event has been triggered).</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>V</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi><mml:mtext>_</mml:mtext><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mi>x</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">{</mml:mo><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mi>e</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mtext>&#x00394;</mml:mtext><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Since the <italic>ET</italic> tensor contains a large number of zero elements, performing operations such as convolution directly would result in a significant amount of meaningless redundant computation. The design of an efficient sparse convolution algorithm that leverages data sparsity will be discussed in detail below.</p></sec>
<sec>
<title>3.2 Valid sub-convolution detection</title>
<p>Convolutional neural networks (CNNs) are an important method of information processing in robotic vision, with the convolution operator being the core of the network. The convolution operation with an image is performed by convolving the kernel with data from a specific window in the input image, which is then traversed across the image using a sliding window. Due to the sparsity of the input image data, the sub-convolution operation at specific windows may contain a certain number of zero computations. When a sub-convolution still contains valid computations, it is called an valid sub-convolution. When a sub-convolution contains only invalid zero computations, it is called an invalid sub-convolution. We denote the index set of valid sub-convolution as <italic>Valid</italic>_<italic>subconv</italic>. valid sub-convolutions are operations that have an impact on the inference results and cannot be ignored. The challenge in achieving efficient sparse convolution lies in how to organize and transform valid sub-convolutions into efficient computations. Before this, it is necessary to first detect the specific locations of the valid sub-convolutions. Thanks to the efficiency of the event-based camera&#x00027;s raw data representation, we can easily obtain the positions of valid pixels in the <italic>ET</italic> tensor, denoted as <italic>Valid</italic>_<italic>pix</italic>, as described in <xref ref-type="disp-formula" rid="E6">Equation 6</xref>. There is a strong correspondence between the positions of valid pixels and the positions of valid sub-convolutions. A window containing an valid sub-convolution must necessarily include valid pixels, and each valid pixel corresponds to a specific valid sub-convolution. Therefore, the locations of the valid sub-convolutions can be directly inferred from <italic>Valid</italic>_<italic>pix</italic>. Specifically, except at the image edges, each valid pixel corresponds to K sub-convolution operations, where K is the size of the sub-convolution window. By matching this pixel with the elements at different positions in the convolution kernel, we can determine the positions of the K sub-convolutions. At the image edges, the number of sub-convolution operations corresponding to each pixel will be fewer than K, and additional checks are required to verify whether the positions of the sub-convolutions are correct and valid. Since the positions of all valid pixels are recorded in <italic>Valid</italic>_<italic>pix</italic>, we can traverse <italic>Valid</italic>_<italic>pix</italic> and take the union of all the positions of the valid sub-convolutions to obtain the complete set of valid sub-convolution positions. This approach greatly improves time efficiency compared to the method of detecting valid sub-convolutions by directly traversing according to their definition.</p></sec>
<sec>
<title>3.3 Sparse Im2col</title>
<p>The convolution operation with an image is essentially a series of logically parallel multiplication and addition operations, forming a very regular computational structure. In practice, however, this often transforms into a more computer-friendly format for parallel computation, specifically general matrix multiplication (GEMM). By converting the convolution operation into GEMM form, the computation speed can be significantly accelerated. To facilitate the introduction of the computational organization involved in this invention, let&#x00027;s first explain the general principle of conventional dense convolution operations. In typical image convolution, the convolution kernel is unfolded into a row vector, and the convolution kernel vectors of different channels form a convolution kernel matrix. The image data in the convolution operation is unfolded into column vectors, and the combination of data from different channels and batches forms an image matrix. By multiplying the resulting convolution kernel matrix with the image matrix, a single general matrix multiplication operation is performed, yielding the result of the image convolution operation (though the element ordering may differ slightly). The overall computation process is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. Since typical dense image convolution operations do not account for invalid calculations, i.e., multiplication and addition operations involving zero elements, processing event-based camera data in this way would waste considerable computational resources, slowing down the operation time.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Conventional dense convolution is usually transformed into general matrix multiplication (GEMM) when executed on actual chips. This transformation helps maintain computational continuity and significantly improves processing speed compared to a naive implementation that strictly follows the definition. It can also be observed that when the input is a sparse tensor, a significant number of redundant computations occur in the GEMM operation. If these redundant computations are eliminated, the overall computation speed can be further improved. The asterisk (&#x0002A;) denotes the &#x0201C;convolution operator,&#x0201D; that is, the convolution in convolutional neural networks.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1537673-g0003.tif"/>
</fig>
<p>This paper introduces an innovation in the image matrix processing step. By removing invalid computation data, the remaining valid operations are transformed into general matrix multiplication (GEMM) operations to achieve efficient computation. The specific approach is as follows: First, for each valid sub-convolution, following the general image convolution method, the image data of one channel in the convolution window is unfolded into a column vector in a row-major order. Then, the column vectors obtained from different channels are concatenated by columns to form the column vector corresponding to the current valid sub-convolution. Next, all the valid sub-convolution operations are traversed in the order defined in <italic>Valid</italic>_<italic>subconv</italic>, and the above steps are repeated to convert all the image data required for valid operations into several column vectors. Finally, all the column vectors corresponding to the valid sub-convolutions are placed into adjacent memory spaces and concatenated by rows to obtain an image matrix. This matrix is then used for general matrix multiplication. In typical dense image convolution operations, the process of converting the image into a matrix for GEMM is called im2col. Therefore, in this paper, the process described above is referred to as sparse im2col. As for processing the convolution kernel matrix, it remains consistent with the general image convolution algorithm. Compared to the im2col in typical image convolution, sparse im2col does not require traversing the entire input tensor directly. It only needs to traverse the smaller, valid convolution portion, bringing two main benefits. On one hand, the computational load for the conversion process is greatly reduced, enabling faster conversion from convolution to GEMM. On the other hand, the resulting matrix is smaller, reducing the computer&#x00027;s memory consumption, and a smaller matrix also means a significant reduction in the computational load for subsequent matrix multiplication operations.</p>
<p><xref ref-type="fig" rid="F4">Figure 4</xref> provides a visual representation of the process described above. For simplicity, only the single-channel convolution is shown, though the actual process is more complex. In the case of typical dense image convolution operations, the matrix size corresponding to the convolution kernel is [<italic>channel</italic>_<italic>out, kernel</italic>_<italic>size</italic> &#x000D7; <italic>kernel</italic>_<italic>size</italic> &#x000D7; <italic>channel</italic>_<italic>in</italic>], and the matrix size corresponding to the image data is [<italic>kernel</italic>_<italic>size</italic> &#x000D7; <italic>kernel</italic>_<italic>size</italic> &#x000D7; <italic>channel</italic>_<italic>in, Valid</italic>_<italic>subconv</italic>]. In general image convolution operations, the matrix size corresponding to the image data is [<italic>kernel</italic>_<italic>size</italic> &#x000D7; <italic>kernel</italic>_<italic>size</italic> &#x000D7; <italic>channel</italic>_<italic>in, N</italic>&#x000D7;<italic>h</italic>_<italic>out</italic> &#x000D7; <italic>w</italic>_<italic>out</italic>]. Therefore, the computational cost ratio of the proposed oprator to that of standard image convolution can be calculated as &#x003B7; &#x0003D; <italic>Valid</italic>_<italic>subconv</italic>/(<italic>N</italic> &#x000D7; <italic>h</italic>_<italic>out</italic> &#x000D7; <italic>w</italic>_<italic>out</italic>). That is the key reason behind the high efficiency of the method. This paper examines the computational sparsity at the sub-convolution level, retains only the valid sub-convolutions, and organizes the remaining valid sub-convolutions into general matrix multiplication operations. This approach not only greatly reduces the computational load but also ensures hardware friendliness in computation.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>We first eliminate redundant computations in the GEMM to reduce the overall computational cost. Then, we organize the remaining valid operations into a GEMM format again to preserve computational continuity. The combination of eliminating redundant computations and preserving computational continuity is the key distinction that sets our method apart from traditional dense convolutions and other sparse convolution approaches. The asterisk (&#x0002A;) denotes the &#x0201C;convolution operator,&#x0201D; that is, the convolution in convolutional neural networks.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1537673-g0004.tif"/>
</fig>
</sec>
<sec>
<title>3.4 Converting the result as regular convolution</title>
<p>The general image convolution operation completed by general matrix multiplication yields a matrix product result <italic>Pd</italic>, which is equivalent to the theoretical convolution result. However, the general matrix multiplication operation completed in section above, which focuses only on the non-zero elements, results in a product <italic>Ps</italic> that differs from the typical image convolution operation. It is a subset of the full convolution result that contains only the valid information. To be compatible and adaptable with other computational components in modern neural networks, such as max pooling and batch normalization, it is necessary to further transform <italic>Ps</italic> into a form identical to that of a typical image convolution. The specific approach is as follows: first, fill the corresponding positions in the resulting product matrix <italic>Ps</italic> with zero elements, expanding it to the same size and shape as the matrix multiplication result <italic>Pd</italic> in typical image convolution. This can be achieved by utilizing <italic>Valid</italic>_<italic>subconv</italic> again. Based on the above description, the following relationship exists between <italic>Ps</italic> and <italic>Pd</italic>:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>V</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi><mml:mtext>_</mml:mtext><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>b</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="false">}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>That is, the result obtained by combining the columns of <italic>Pd</italic> at the index positions of the elements in <italic>Valid</italic>_<italic>subconv</italic> is <italic>Ps</italic>. This can be easily explained, as in matrix multiplication, the columns of the product correspond one-to-one with the columns of the operand matrix. Therefore, by using the above relationship in reverse, we can remap the elements in <italic>Ps</italic> to new positions based on <italic>Valid</italic>_<italic>subconv</italic>, and fill the complement positions of <italic>Valid</italic>_<italic>subconv</italic> with zero to obtain the desired result, denoted as <italic>Pds</italic>. Next, <italic>Pds</italic> needs to be further transformed into the format of a typical image convolution. The result <italic>Pds</italic>, obtained from general matrix multiplication, has a data layout format of [C &#x000D7; N &#x000D7; H &#x000D7; W], whereas modern neural networks typically use the [N &#x000D7; C &#x000D7; H &#x000D7; W] format. Thus, by swapping the axes, the final required image convolution result can be obtained.</p></sec></sec>
<sec sec-type="results" id="s4">
<title>4 Results</title>
<p>To verify the correctness and effectiveness of the proposed method, we designed a neural network with a single convolution layer and evaluated the proposed sparse convolution operator from three dimensions: computational accuracy, computational complexity, and actual inference time. For the input data, we selected the N-Caltech101 dataset (Orchard et al., <xref ref-type="bibr" rid="B17">2015a</xref>), which is derived from real event-based camera recordings. This dataset captures images from the Caltech101 dataset using an event-based camera, generating corresponding event data. It contains 8,246 samples across 101 categories, with each sample comprising 300 ms of event-based camera data. The image resolution is 180 &#x000D7; 240. The experimental platform utilized an Intel i7-12700KF CPU as the computing unit, with 16GB of memory and running Ubuntu 20 as the operating system. The code was compiled using gcc with the -O3 optimization option enabled.</p>
<p>The primary factor influencing the inference performance of a sparse convolution operator is the sparsity level of the input data. In theory, the higher the sparsity, the more evident the advantages of the sparse convolution algorithm. In the extreme case where the sparsity level is zero, the algorithm degenerates into a dense convolution. To investigate the performance of the proposed sparse convolution operator under varying sparsity levels, we generated input data with different sparsity by extracting event data from time windows of different lengths within the samples.</p>
<p>To more clearly demonstrate the advantages of the proposed sparse convolution operator, we conducted comparative experiments with both dense convolution and the classical SCN sparse convolution method. For the dense convolution implementation, we selected the standard dense convolution operator from Intel&#x00027;s MKL library, which offers the highest inference efficiency for convolutional neural networks on Intel CPU products. This is due to extensive engineering optimizations specifically tailored to the characteristics of Intel CPUs.</p>
<p>In the N-Caltech101 dataset, each 300 ms sample is recorded by moving the camera in three different directions relative to a static image, with each movement lasting 100 ms. Therefore, the first 100 ms of data can be considered representative of the entire sample. To generate data with varying sparsity levels, we extracted time windows of different sizes from this 100 ms segment, starting from 0 ms, with a time step of 1 ms, resulting in 100 sets of event-based camera data with different densities. The raw event data is then converted into a tensor following the previously described method, yielding an input tensor with the shape [8, 2, 180, 240]. When the time window is set to 100 ms, the maximum density of the input tensor reaches approximately 3%.</p>
<p>First, we validate the computational accuracy. Theoretically, the proposed convolution operator only eliminates redundant multiplications and additions involving zeros, so its results should be identical to those of standard dense convolution. We randomly initialize the weights for the dense convolution and assign the same weights to the proposed sparse convolution operator. Using PyTorch&#x00027;s allclose function, we verify whether the results are identical. The relative and absolute tolerance values are set to 10<sup>&#x02212;3</sup> and 10<sup>&#x02212;5</sup>, respectively. The statistical results show that the outputs of both methods fall within the tolerance range, consistent with theoretical expectations, confirming the correctness and reliability of the proposed sparse convolution operator.</p>
<p>Next, we conduct a comparative experiment on computational cost. Since the primary operations in the convolution process are multiplications and additions, we use the number of multiply-add operations as the basis for comparison. To clearly illustrate the differences in computational cost between the operators, we normalize the computational cost using the dense convolution as the baseline. The results are shown in the <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Comparison of the computational complexity of different convolution operators. The computational complexity of the method in this paper is the same as that of the SCN method. The line represents the average complexity, and the shaded envelope represents the standard deviation of the complexity.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1537673-g0005.tif"/>
</fig>
<p>It can be observed that the proposed sparse convolution operator has the same computational cost as the classical SCN sparse convolution algorithm, both of which are significantly lower than that of dense convolution. Even when the time window is set to 100 ms, where the input tensor contains the majority of the effective information, the computational cost is only about 25% of that of dense convolution.</p>
<p>Next, we conducted an experiment to measure the actual inference time. The 100 input tensors mentioned above were fed into the standard dense convolution operator, the SCN sparse convolution operator, and the proposed sparse convolution operator, respectively. By evaluating the samples from the N-Caltech101 test set, we obtained the inference performance of the three operators under different input sparsity levels, as shown in <xref ref-type="fig" rid="F6">Figure 6</xref>.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Comparison of the actual inference speed of different convolution operators. For each time window setting, the N-Caltech101 test set is traversed once to ensure multiple experiments for each sparsity level. The line represents the average inference time, and the shaded envelope represents the standard deviation of the inference time.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1537673-g0006.tif"/>
</fig>
<p>The experiment shows that as sparsity increases, the inference time of the SCN algorithm quickly surpasses that of dense convolution, performing better only when the data density is extremely low. However, once the density rises to around 1%&#x02013;corresponding to a time window of approximately 10 ms&#x02013;the inference speed of SCN falls behind dense convolution, limiting its practical applicability. In contrast, the proposed sparse convolution operator maintains competitive performance until the sparsity decreases to around 2.2%, with a corresponding time window of approximately 65 ms. Moreover, the proposed operator consistently outperforms the SCN convolution operator in terms of inference time, demonstrating superior inference efficiency.</p>
<p>In summary, the experimental results demonstrate that the proposed sparse convolution operator not only significantly reduces computational cost but also effectively translates this reduction into shorter inference time, while maintaining the accuracy of the inference results. This achieves efficient and accurate sparse convolution inference.</p></sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>The sparse convolution operator proposed in this paper strikes a good balance between computational complexity and actual inference time, breaking the previous situation where sparse convolution operators designed for event-based camera data achieved significant computational advantages but showed no clear benefits in practical inference. This is mainly because the sparse convolution operator proposed in this paper not only ensures sparsity but also aligns with the computational continuity characteristics of computing devices, generating only minimal additional computational overhead during the organization of operations.</p>
<p>Furthermore, the method proposed in this paper has an additional advantage over other sparse convolution algorithms, such as SCN: the input and output of the sparse convolution operator proposed here are consistent with conventional convolution operators, allowing for interchangeability without the need for specialized conversion between sparse and dense representations. This provides more flexibility in building neural networks.</p>
<p>However, the sparse convolution operator proposed in this paper also has certain drawbacks. Specifically, as the sparsity of the input data decreases, its computational efficiency advantage over conventional convolution operators gradually diminishes and eventually disappears. This is mainly because conventional convolution operators also utilize a large number of engineering optimization techniques. In contrast, the sparse convolution operator presented here is a preliminary prototype. But this also highlights the superiority of the idea of sparse convolution approach proposed in this paper. The author believes that the method proposed in this paper is not in conflict with other optimization techniques in conventional convolution operators, and that the performance advantage will become even greater once these engineering optimizations are incorporated in the future.</p></sec>
<sec sec-type="conclusions" id="s6">
<title>6 Conclusion</title>
<p>In this paper, we present an efficient sparse convolution operator tailored for event-based cameras and validate its advantages through extensive experiments. Unlike traditional methods, our approach harnesses matrix multiplication to maintain operational continuity, effectively transforming reduced computational complexity into a substantial decrease in inference time. Remarkably, our operator reduces the computational workload by nearly 90% while nearly doubling processing speed, all while preserving the accuracy of dense convolution operators.</p>
<p>Thus far, our research has primarily focused on optimizing the implementation of a single convolution operator. Given that our operator maintains compatibility with conventional convolution operators in terms of input/output formats and computational processes, we will next extend its application to complete convolutional neural networks, which will enhance robotic perception and responsiveness in high-speed, emergency scenarios, providing a robust safety guarantee for large-scale real-world applications.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>SZ: Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. FZ: Funding acquisition, Resources, Writing &#x02013; review &#x00026; editing. XW: Writing &#x02013; review &#x00026; editing. ML: Writing &#x02013; review &#x00026; editing. WG: Visualization, Writing &#x02013; review &#x00026; editing. PW: Writing &#x02013; review &#x00026; editing. XL: Writing &#x02013; review &#x00026; editing. LS: Supervision, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported in part by the National Key R&#x00026;D Program of China (2022YFB4601800), National Natural Science Foundation of China (U2013602, 52075115, 51521003, 61911530250, and 52105307), Self-Planned Task (SKLRS202001B and SKLRS202110B) of State Key Laboratory of Robotics and System (HIT), Shenzhen Science and Technology Research and Development Foundation (JCYJ20190813171009236), Basic Scientific Research of Technology (JCKY2020603C009), School Enterprise Joint R&#x00026;D Center for Cemented Carbide Cutting Tools of Shenzhen Polytechnic University (602431003PQ), and The Key Talent Project of Gansu Province.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p></sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benosman</surname> <given-names>R.</given-names></name> <name><surname>Clercq</surname> <given-names>C.</given-names></name> <name><surname>Lagorce</surname> <given-names>X.</given-names></name> <name><surname>Ieng</surname> <given-names>S.-H.</given-names></name> <name><surname>Bartolozzi</surname> <given-names>C.</given-names></name></person-group> (<year>2013</year>). <article-title>Event-based visual flow</article-title>. <source>IEEE trans. Neural Netw. Learn. Syst</source>. <volume>25</volume>, <fpage>407</fpage>&#x02013;<lpage>417</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2013.2273537</pub-id><pub-id pub-id-type="pmid">24807038</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bing</surname> <given-names>Z.</given-names></name> <name><surname>Meschede</surname> <given-names>C.</given-names></name> <name><surname>Huang</surname> <given-names>K.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name> <name><surname>Rohrbein</surname> <given-names>F.</given-names></name> <name><surname>Akl</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;End to end learning of spiking neural network based on R-STDP for a lane keeping vehicle,&#x0201D;</article-title> in <source>2018 IEEE International Conference on Robotics and Automation (ICRA)</source> (<publisher-loc>Brisbane, QLD</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4725</fpage>&#x02013;<lpage>4732</lpage>.<pub-id pub-id-type="pmid">31526952</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brosch</surname> <given-names>T.</given-names></name> <name><surname>Tschechne</surname> <given-names>S.</given-names></name> <name><surname>Neumann</surname> <given-names>H.</given-names></name></person-group> (<year>2015</year>). <article-title>On event-based optical flow detection</article-title>. <source>Front. Neurosci</source>. <volume>9</volume>:<fpage>137</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2015.00137</pub-id><pub-id pub-id-type="pmid">25941470</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Zha</surname> <given-names>F.</given-names></name> <name><surname>Guo</surname> <given-names>W.</given-names></name> <name><surname>Du</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>A2G: Leveraging intuitive physics for force-efficient robotic grasping</article-title>. <source>IEEE Robot. Automat. Lett</source>. <volume>9</volume>, <fpage>6376</fpage>&#x02013;<lpage>6383</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2024.3401675</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cordone</surname> <given-names>L.</given-names></name> <name><surname>Miramond</surname> <given-names>B.</given-names></name> <name><surname>Ferrante</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Learning from event cameras with sparse spiking convolutional neural networks,&#x0201D;</article-title> in <source>2021 International Joint Conference on Neural Networks (IJCNN)</source> (<publisher-loc>Shenzhen</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>866</fpage>&#x02013;<lpage>880</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gallego</surname> <given-names>G.</given-names></name> <name><surname>Delbr&#x000FC;ck</surname> <given-names>T.</given-names></name> <name><surname>Orchard</surname> <given-names>G.</given-names></name> <name><surname>Bartolozzi</surname> <given-names>C.</given-names></name> <name><surname>Taba</surname> <given-names>B.</given-names></name> <name><surname>Censi</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Event-based vision: a survey</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>44</volume>, <fpage>154</fpage>&#x02013;<lpage>180</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2020.3008413</pub-id><pub-id pub-id-type="pmid">32750812</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Graham</surname> <given-names>B.</given-names></name> <name><surname>Engelcke</surname> <given-names>M.</given-names></name> <name><surname>Van Der Maaten</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;3D semantic segmentation with submanifold sparse convolutional networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision And Pattern Recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>9224</fpage>&#x02013;<lpage>9232</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Graham</surname> <given-names>B.</given-names></name> <name><surname>Van der Maaten</surname> <given-names>L.</given-names></name></person-group> (<year>2017</year>). <article-title>Submanifold sparse convolutional networks</article-title>. <source>arXiv</source> [preprint] arXiv:1706.01307. <pub-id pub-id-type="doi">10.1109/CVPR.2018.00961</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>H.</given-names></name> <name><surname>Zheng</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Gao</surname> <given-names>Z.</given-names></name> <name><surname>Zhao</surname> <given-names>S.</given-names></name></person-group> (<year>2024</year>). <article-title>Global-local mav detection under challenging conditions based on appearance and motion</article-title>. <source>IEEE Trans. Intellig. Transp. Syst</source>. <volume>25</volume>, <fpage>12005</fpage>&#x02013;<lpage>12017</lpage>. <pub-id pub-id-type="doi">10.1109/TITS.2024.3381174</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>Z.</given-names></name> <name><surname>Bing</surname> <given-names>Z.</given-names></name> <name><surname>Huang</surname> <given-names>K.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name> <name><surname>Cheng</surname> <given-names>L.</given-names></name> <name><surname>Knoll</surname> <given-names>A.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Event-based target tracking control for a snake robot using a dynamic vision sensor,&#x0201D;</article-title> in <source>Neural Information Processing: 24th International Conference, ICONIP 2017</source> (<publisher-loc>Guangzhou</publisher-loc>: <publisher-name>Springer</publisher-name>).</citation>
</ref>
<ref id="B11">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>Z.</given-names></name> <name><surname>Xia</surname> <given-names>P.</given-names></name> <name><surname>Huang</surname> <given-names>K.</given-names></name> <name><surname>Stechele</surname> <given-names>W.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name> <name><surname>Bing</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Mixed frame-/event-driven fast pedestrian detection,&#x0201D;</article-title> in <source>2019 International Conference on Robotics and Automation (ICRA)</source> (<publisher-loc>Montreal, QC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>8332</fpage>&#x02013;<lpage>8338</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>M.</given-names></name> <name><surname>Delbruck</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). <source>Adaptive Time-Slice Block-Matching Optical Flow Algorithm for Dynamic Vision Sensors</source>. <publisher-loc>Glasgow</publisher-loc>: <publisher-name>BMVC</publisher-name>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meng</surname> <given-names>Y.</given-names></name> <name><surname>Bing</surname> <given-names>Z.</given-names></name> <name><surname>Yao</surname> <given-names>X.</given-names></name> <name><surname>Chen</surname> <given-names>K.</given-names></name> <name><surname>Huang</surname> <given-names>K.</given-names></name> <name><surname>Gao</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Preserving and combining knowledge in robotic lifelong reinforcement learning</article-title>. <source>Nature Mach. Intellig</source>. <volume>7</volume>, <fpage>256</fpage>&#x02013;<lpage>269</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-025-00983-2</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Messikommer</surname> <given-names>N.</given-names></name> <name><surname>Gehrig</surname> <given-names>D.</given-names></name> <name><surname>Loquercio</surname> <given-names>A.</given-names></name> <name><surname>Scaramuzza</surname> <given-names>D.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Event-based asynchronous sparse convolutional networks,&#x0201D;</article-title> in <source>Computer Vision-ECCV 2020: 16th European Conference</source> (<publisher-loc>Glasgow</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>23</fpage>&#x02013;<lpage>28</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Miao</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name> <name><surname>Ning</surname> <given-names>X.</given-names></name> <name><surname>Zi</surname> <given-names>Y.</given-names></name> <name><surname>Ren</surname> <given-names>K.</given-names></name> <name><surname>Bing</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Neuromorphic vision datasets for pedestrian detection, action recognition, and fall detection</article-title>. <source>Front. Neurorobot</source>. <volume>13</volume>:<fpage>38</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2019.00038</pub-id><pub-id pub-id-type="pmid">31275128</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Munda</surname> <given-names>G.</given-names></name> <name><surname>Reinbacher</surname> <given-names>C.</given-names></name> <name><surname>Pock</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). <article-title>Real-time intensity-image reconstruction for event cameras using manifold regularisation</article-title>. <source>Int. J. Comput. Vis</source>. <volume>126</volume>, <fpage>1381</fpage>&#x02013;<lpage>1393</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-018-1106-2</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Orchard</surname> <given-names>G.</given-names></name> <name><surname>Jayawant</surname> <given-names>A.</given-names></name> <name><surname>Cohen</surname> <given-names>G. K.</given-names></name> <name><surname>Thakor</surname> <given-names>N.</given-names></name></person-group> (<year>2015a</year>). <article-title>Converting static image datasets to spiking neuromorphic datasets using saccades</article-title>. <source>Front. Neurosci</source>. <volume>9</volume>:<fpage>437</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2015.00437</pub-id><pub-id pub-id-type="pmid">26635513</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Orchard</surname> <given-names>G.</given-names></name> <name><surname>Meyer</surname> <given-names>C.</given-names></name> <name><surname>Etienne-Cummings</surname> <given-names>R.</given-names></name> <name><surname>Posch</surname> <given-names>C.</given-names></name> <name><surname>Thakor</surname> <given-names>N.</given-names></name> <name><surname>Benosman</surname> <given-names>R.</given-names></name></person-group> (<year>2015b</year>). <article-title>HFIRST: a temporal approach to object recognition</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>37</volume>, <fpage>2028</fpage>&#x02013;<lpage>2040</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2015.2392947</pub-id><pub-id pub-id-type="pmid">26353184</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Parger</surname> <given-names>M.</given-names></name> <name><surname>Tang</surname> <given-names>C.</given-names></name> <name><surname>Neff</surname> <given-names>T.</given-names></name> <name><surname>Twigg</surname> <given-names>C. D.</given-names></name> <name><surname>Keskin</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2022a</year>). <article-title>Motiondeltacnn: sparse cnn inference of frame differences in moving camera videos</article-title>. <source>arXiv</source> [preprint] arXiv:2210.09887. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01217</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Parger</surname> <given-names>M.</given-names></name> <name><surname>Tang</surname> <given-names>C.</given-names></name> <name><surname>Twigg</surname> <given-names>C. D.</given-names></name> <name><surname>Keskin</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>R.</given-names></name> <name><surname>Steinberger</surname> <given-names>M.</given-names></name></person-group> (<year>2022b</year>). <article-title>&#x0201C;DeltaCNN: end-to-end cnn inference of sparse frame differences in videos,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>. <fpage>12497</fpage>&#x02013;<lpage>12506</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qian</surname> <given-names>R.</given-names></name> <name><surname>Lai</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name></person-group> (<year>2022</year>). <article-title>3D object detection for autonomous driving: a survey</article-title>. <source>Pattern Recognit</source>. <volume>130</volume>:<fpage>108796</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2022.108796</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rebecq</surname> <given-names>H.</given-names></name> <name><surname>Ranftl</surname> <given-names>R.</given-names></name> <name><surname>Koltun</surname> <given-names>V.</given-names></name> <name><surname>Scaramuzza</surname> <given-names>D.</given-names></name></person-group> (<year>2019</year>). <article-title>High speed and high dynamic range video with an event camera</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>43</volume>, <fpage>1964</fpage>&#x02013;<lpage>1980</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2019.2963386</pub-id><pub-id pub-id-type="pmid">31902754</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schaefer</surname> <given-names>S.</given-names></name> <name><surname>Gehrig</surname> <given-names>D.</given-names></name> <name><surname>Scaramuzza</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;AEGNN: Asynchronous event-based graph neural networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, <fpage>12371</fpage>&#x02013;<lpage>12381</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sorin</surname> <given-names>D.</given-names></name> <name><surname>Hill</surname> <given-names>M.</given-names></name> <name><surname>Wood</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <source>A Primer on Memory Consistency and Cache Coherence</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature</publisher-name>.</citation>
</ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Chao</surname> <given-names>W.-L.</given-names></name> <name><surname>Garg</surname> <given-names>D.</given-names></name> <name><surname>Hariharan</surname> <given-names>B.</given-names></name> <name><surname>Campbell</surname> <given-names>M.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Pseudo-lidar from visual depth estimation: Bridging the gap in 3D object detection for autonomous driving,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Long Beach, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>8445</fpage>&#x02013;<lpage>8453</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>D.-H.</given-names></name> <name><surname>Lu</surname> <given-names>X.-T.</given-names></name> <name><surname>Yang</surname> <given-names>F.</given-names></name> <name><surname>Yao</surname> <given-names>M.</given-names></name> <name><surname>Dong</surname> <given-names>W.-S.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Efficient visual recognition: a survey on recent advances and brain-inspired methodologies</article-title>. <source>Mach. Intellig. Res</source>. <volume>19</volume>, <fpage>366</fpage>&#x02013;<lpage>411</lpage>. <pub-id pub-id-type="doi">10.1007/s11633-022-1340-5</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>Y.</given-names></name> <name><surname>Mao</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>B.</given-names></name></person-group> (<year>2018</year>). <article-title>Second: Sparsely embedded convolutional detection</article-title>. <source>Sensors</source> <volume>18</volume>:<fpage>3337</fpage>. <pub-id pub-id-type="doi">10.3390/s18103337</pub-id><pub-id pub-id-type="pmid">30301196</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>B.</given-names></name> <name><surname>Tang</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>S.-S.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Autonomous driving digital twin empowered design automation: an industry perspective,&#x0201D;</article-title> in <source>2023 60th ACM/IEEE Design Automation Conference (DAC)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1</fpage>&#x02013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>A. Z.</given-names></name> <name><surname>Yuan</surname> <given-names>L.</given-names></name> <name><surname>Chaney</surname> <given-names>K.</given-names></name> <name><surname>Daniilidis</surname> <given-names>K.</given-names></name></person-group> (<year>2018</year>). <article-title>EV-FLOWNET: Self-supervised optical flow estimation for event-based cameras</article-title>. <source>arXiv</source> [preprint] arXiv:1802.06898. <pub-id pub-id-type="doi">10.15607/RSS.2018.XIV.062</pub-id></citation>
</ref>
</ref-list>
</back>
</article> 