<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title-group>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1541017</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2025.1541017</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>VIO-GO: optimizing event-based SLAM parameters for robust performance in high dynamic range scenarios</article-title>
<alt-title alt-title-type="left-running-head">Sakhrieh et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2025.1541017">10.3389/frobt.2025.1541017</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Sakhrieh</surname>
<given-names>Saber</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Singh</surname>
<given-names>Abhilasha</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Mounsef&#x2009;</surname>
<given-names>Jinane</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2782556"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Arain&#x2009;</surname>
<given-names>Bilal</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Maalouf&#x2009;</surname>
<given-names>Noel</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2816264"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution>Electrical Engineering and Computing Sciences Department, Rochester Institute of Technology</institution>, <city>Dubai</city>, <country country="AE">United Arab Emirates</country>
</aff>
<aff id="aff2">
<label>2</label>
<institution>Department of Computer Engineering, University of Sharjah</institution>, <city>Sharjah</city>, <country country="AE">United Arab Emirates</country>
</aff>
<aff id="aff3">
<label>3</label>
<institution>Electrical and Computer Engineering Department, Lebanese American University</institution>, <city>Byblos</city>, <country country="LB">Lebanon</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Jinane Mounsef, <email xlink:href="jmbcad@rit.edu">jmbcad@rit.edu</email>
</corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-09-18">
<day>18</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1541017</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>12</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>01</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Sakhrieh, Singh, Mounsef&#x2009;, Arain&#x2009; and Maalouf&#x2009;.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Sakhrieh, Singh, Mounsef&#x2009;, Arain&#x2009; and Maalouf&#x2009;</copyright-holder>
<license>
<ali:license_ref start_date="2025-09-18">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>This paper addresses a critical challenge in Industry 4.0 robotics by enhancing Visual Inertial Odometry (VIO) systems to operate effectively in dynamic and low-light industrial environments, which are common in sectors like warehousing, logistics, and manufacturing. Inspired by biological sensing mechanisms, we integrate bio-inspired event cameras to improve state estimation systems performance in both dynamic and low-light conditions, enabling reliable localization and mapping. The proposed state estimation framework integrates events, conventional video frames, and inertial data to achieve reliable and precise localization with specific emphasis on real-world challenges posed by high-speed and cluttered settings typical in Industry 4.0. Despite advancements in event-based sensing, there is a noteworthy gap in optimizing Event Simultaneous Localization and Mapping (SLAM) parameters for practical applications. To address this, we introduce a novel VIO-Gradient-based Optimization (VIO-GO) method that employs Batch Gradient Descent (BGD) for efficient parameter tuning. This automated approach determines optimal parameters for Event SLAM algorithms by using motion-compensated images to represent event data. Experimental validation on the Event Camera Dataset shows a remarkable 60% improvement in Mean Position Error (MPE) over fixed-parameter methods. Our results demonstrate that VIO-GO consistently identifies optimal parameters, enabling precise VIO performance in complex, dynamic scenarios essential for Industry 4.0 applications. Additionally, as parameter complexity scales, VIO-GO achieves a 24% reduction in MPE when using the most comprehensive parameter set (VIO-GO8) compared to a minimal set (VIO-GO2), highlighting the method&#x2019;s scalability and robustness for adaptive robotic systems in challenging industrial environments.</p>
</abstract>
<kwd-group>
<kwd>visual inertial odometry</kwd>
<kwd>event SLAM</kwd>
<kwd>batch gradient descent</kwd>
<kwd>optimization</kwd>
<kwd>edge image</kwd>
<kwd>dynamic and low-light environments</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declare that no financial support was received for the research and/or publication of this article.</funding-statement>
</funding-group>
<counts>
<fig-count count="9"/>
<table-count count="5"/>
<equation-count count="11"/>
<ref-count count="35"/>
<page-count count="18"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Industrial Robotics and Automation</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>SLAM is a key technology in the autonomous navigation of robots, serving as a fundamental element for the operation of autonomous vehicles (<xref ref-type="bibr" rid="B29">Sahili et al., 2023</xref>). Over the past 2 decades, research in SLAM and Visual Odometry (VO), using cameras either independently or in conjunction with inertial sensors, has led to highly accurate and robust systems that continually improve in performance (<xref ref-type="bibr" rid="B3">Campos et al., 2021</xref>).</p>
<p>As the complexity of autonomous applications grows, new challenges require innovative VO and SLAM solutions that deliver precision and reliability in increasingly dynamic scenarios. Standard cameras, while effective in certain conditions, encounter difficulties in low-light environments or during rapid movement due to motion blur and constrained frame rates. Event cameras, asynchronous visual sensors that address these issues, have emerged as a promising alternative (<xref ref-type="bibr" rid="B11">Hadviger et al., 2023</xref>). Event-based SLAM offers distinct advantages by eliminating motion blur, supporting high dynamic range (HDR), and operating at higher frame rates. However, event cameras may struggle in scenarios with minimal relative motion, such as in stationary states, where standard cameras excel in offering instantaneous and comprehensive environmental data, particularly in low-speed, well-lit conditions (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>).</p>
<p>This complementary nature suggests a hybrid approach, combining event and standard cameras with an inertial measurement unit (IMU) to produce a reliable and precise VIO framework (<xref ref-type="bibr" rid="B6">Chen et al., 2023</xref>). By compensating for each sensor&#x2019;s limitations, this integrated framework is versatile and adaptable to a broad range of environmental conditions and movement patterns. One such example is Ultimate SLAM (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>), which integrates traditional cameras, event frames, and IMU data to provide reliable state estimation even in challenging scenarios.</p>
<p>As event cameras transform the way visual data is captured, new methodologies are needed to handle and interpret this unique data effectively (<xref ref-type="bibr" rid="B29">Sahili et al., 2023</xref>). A major limitation involves adapting to the asynchronous and sparse output from event cameras, contrasting with the dense, synchronous images produced by conventional cameras. Consequently, traditional vision algorithms developed for frame-based image sequences cannot be directly applied to event data (<xref ref-type="bibr" rid="B7">Gallego et al., 2022</xref>).</p>
<p>To bridge this gap, several techniques have been presented to transform asynchronous event data into a synchronous format (<xref ref-type="bibr" rid="B10">Guan et al., 2023</xref>). Some methods directly process raw event streams without accumulating frames (<xref ref-type="bibr" rid="B1">Alzugaray and Chli, 2018</xref>). Others use learning-based methods to create intensity images from events (<xref ref-type="bibr" rid="B8">Gehrig et al., 2020</xref>). Another approach is the generation of motion-compensated event or edge images by grouping events within specific spatial and temporal windows, highlighting scene edges and providing a structured visual representation of event data (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>; <xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>). However, this method presents challenges, requiring substantial parameter adjustments specific to each environment, which becomes burdensome when transitioning across diverse scenarios (<xref ref-type="bibr" rid="B12">Huang et al., 2024</xref>). This dependency on manual tuning creates a bottleneck for VIO systems, which must exhibit robust performance in unfamiliar environments where the event count varies widely. For practical applications, especially within Industry 4.0, where environments are constantly changing, manual parameter adjustments become impractical (<xref ref-type="bibr" rid="B19">Mahlknecht et al., 2022</xref>).</p>
<p>In this paper, we present VIO-GO, a novel approach designed to enhance the performance and robustness of Event SLAM in unknown environments through targeted parameter optimization. Our method focuses on tuning parameters for Visual SLAM systems that use motion-compensated images to represent event data. By integrating event-based SLAM methods with a BGD algorithm, VIO-GO enables iterative refinement of parameters across diverse scenes with varying event generation rates. Unlike conventional motion-compensated image methods, VIO-GO minimizes the need for extensive manual parameter tuning, leading to improved time efficiency. Experimental results indicate that VIO-GO outperforms both fixed-parameter motion-compensated approaches and state-of-the-art EVIO methods, demonstrating superior performance across various dynamic scenes. <xref ref-type="fig" rid="F1">Figure 1</xref> shows a comparison between VIO-GO and Ultimate SLAM using its fine-tuned parameters, highlighting the improvements in performance achieved by VIO-GO.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The estimated trajectory of Ultimate SLAM (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>) and VIO-GO aligned with the ground truth trajectory for the <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>r</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>b</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> sequence in (<xref ref-type="bibr" rid="B22">Mueggler et al., 2017b</xref>). The figures show that VIO-GO produces a precise trajectory with a low absolute position error (APE) value, which is comparable to the trajectory estimated by Ultimate SLAM. <bold>(a)</bold> Estimated trajectory (fine-tuned parameters). <bold>(b)</bold> Estimated trajectory (VIO-GO).</p>
</caption>
<graphic xlink:href="frobt-12-1541017-g001.tif">
<alt-text content-type="machine-generated">Two 3D plots compare estimated trajectories with reference lines. The left plot shows a trajectory with fine-tuned parameters, and the right plot shows a VIO-GO trajectory. Both include insets highlighting specific sections, with a color scale from blue to red indicating values, and dashed lines for reference trajectories. Axes are marked with x, y, z coordinates in meters.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<p>The integration of RGB cameras and inertial sensors has long been foundational in VIO systems. However, recent advancements have seen an increasing shift towards the inclusion of event cameras, marking a pivotal development in the field of Visual SLAM. This literature review is divided into two subsections: Event-Based Visual-Inertial Odometry (EVIO) and Adaptive Parameter Optimization, providing a focused examination of the latest research.</p>
<sec id="s2-1">
<label>2.1</label>
<title>Event-based visual-Inertial odometry (EVIO)</title>
<p>EVIO integrates the high-speed, high-contrast sensitivity of event cameras combined with inertial data from accelerometers and gyroscopes (<xref ref-type="bibr" rid="B6">Chen et al., 2023</xref>). Unlike traditional cameras, event cameras capture changes in a scene at up to 1 MHz, allowing them to handle rapid motion without issues like image blur, temporal aliasing, or saturation under intense lighting. These attributes make event cameras particularly useful for dynamic and low-light environments (<xref ref-type="bibr" rid="B35">Zhu et al., 2017</xref>).</p>
<p>The study in (<xref ref-type="bibr" rid="B21">Mueggler et al., 2017a</xref>) examines the integration of event frames and inertial data using a continuous-time model. However, real-time application remains challenging due to the computational burden of adjusting spline parameters for each incoming event. In a different approach, the authors in (<xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>) propose a real-time event-based VIO pipeline that uses optical flow estimation, grounded in the recent camera pose, scene configuration, and inertial data, to track visual features across multiple frames. These tracked features are subsequently combined with inertial measurements through keyframe-based nonlinear optimization. Similarly, in (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>), the authors integrate event streams, standard frames, and IMU data through nonlinear optimization process, achieving a notable accuracy enhancement of 130% over event-only frames and 85% over standard frames with IMU data. This system also supports real-time integration with a quadrotor. Furthermore, the authors in (<xref ref-type="bibr" rid="B4">Censi and Scaramuzza, 2014</xref>; <xref ref-type="bibr" rid="B14">Kueng et al., 2016</xref>) introduce low-latency, event-based VO techniques that accurately estimate rotation and translation using Dynamic Vision Sensors (DVS) alongside conventional CMOS cameras in natural scenes.</p>
<p>In (<xref ref-type="bibr" rid="B13">Kim et al., 2016</xref>), the authors discuss an event-based 6-degree-of-freedom (6-DoF) VO system, using three decoupled probabilistic filters to estimate the camera&#x2019;s pose, a 3D scene model, and image intensity. However, this approach incurs a significant computational burden, requiring GPU use for real-time performance. In (<xref ref-type="bibr" rid="B35">Zhu et al., 2017</xref>), the EVIO method is presented, where an Extended Kalman Filter (EKF) fuses event data with pre-integrated IMU measurements, demonstrating the potential of event-based VIO for applications like planetary exploration. To further explore this potential (<xref ref-type="bibr" rid="B19">Mahlknecht et al., 2022</xref>), introduce the Event-based Lucas-Kanade Tracking VIO (EKLT-VIO), which integrates an event-based tracker developed by (<xref ref-type="bibr" rid="B8">Gehrig et al., 2020</xref>) in the front-end and a filter-based back-end to conduct VIO in Mars-like environments. Their results show a 32% improvement in MPE under low-light and HDR conditions compared to prior frame-based and event-based VIO techniques, although front-end and back-end parameter selection were not addressed. Expanding upon previous methodologies, PL-EVIO, proposed by (<xref ref-type="bibr" rid="B10">Guan et al., 2023</xref>), tightly integrates event-based point and line features, standard frame point features, and IMU data, thereby providing more geometric restrictions and enhancing robustness. While these advancements mark significant progress in Event SLAM, the literature reveals a gap in adaptive parameter optimization across various scenarios. Our approach aligns closely with methodologies in (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>; <xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>), which use motion-compensated images to represent event data but require significant manual parameter tuning for diverse environments. Our goal is to streamline the process by developing a pipeline capable of automating parameter optimization, thereby enhancing both the practicability and time efficiency of Event SLAM systems.</p>
</sec>
<sec id="s2-2">
<label>2.2</label>
<title>Adaptive parameter optimization</title>
<p>Integrating event cameras into VO/VIO systems presents a major challenge that arises from the asynchronous nature of event streams, which fundamentally differs from synchronous image data. Consequently, many methods designed for traditional image-based cameras cannot be directly applied to event-based systems. To address this gap, various techniques for representing event data have been introduced in the literature (<xref ref-type="bibr" rid="B10">Guan et al., 2023</xref>). Common approaches involve applying conventional feature detection and tracking methods to edge images created from motion-compensated event streams (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>; <xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>). However, these methods often require extensive parameter adjustments to adapt to specific fluctuations in event density, which can impact VIO system performance. Through a review of existing parameter optimization methods, we aim to identify the most effective strategies for enhancing event-based VIO systems and highlight areas where further optimization could improve system adaptability and reliability.</p>
<p>In (<xref ref-type="bibr" rid="B16">Li et al., 2020</xref>), the authors use a Stochastic Gradient Descent (SGD) approach for localization, coupled with scan matching via a 2D LiDAR system. This SGD-based approach enables the localizer to effectively track the robot&#x2019;s state, generating a coherent trajectory of its movements. The technique attained a position error of 0.26 m and a heading error of around 5&#xb0;. In (<xref ref-type="bibr" rid="B32">Torroba et al., 2023</xref>), the authors employ SGD to optimize the evidence lower bound (ELBO) on Gaussian process maps by estimating mini-batches, which allowed real-time performance on large-scale datasets and was successfully tested in a live Autonomous Underwater Vehicle (AUV) mission. Similarly, the authors in (<xref ref-type="bibr" rid="B31">Song et al., 2021</xref>) examine SGD for map classification in SLAM, while (<xref ref-type="bibr" rid="B2">Beomsoo et al., 2021</xref>) implements SGD to refine the policy network within the Proximal Policy Optimization algorithm. These approaches demonstrate the effectiveness of SGD in achieving both accuracy and efficiency for front-end and back-end optimization in dynamic environments.</p>
<p>In (<xref ref-type="bibr" rid="B26">Rebecq et al., 2017a</xref>), the authors use an intermediate representation by accumulating events into an edge-like image, employing a Gradient Descent (GD) approach that simplifies representation by randomly sampling pixels. This technique improves tracker speed and enhances robustness by increasing resilience to occlusions. In another approach, <xref ref-type="bibr" rid="B18">Luo et al. (2019)</xref> introduce a stage-wise SGD algorithm with a selective update mechanism to efficiently select a subset of training images for direct SLAM tracking, ensuring faster convergence.</p>
<p>As discussed, although significant research has focused on optimizing conventional SLAM methods, limited studies have applied GD approaches specifically to optimize Event SLAM parameters. Given the potential benefits, this work adopts the GD approach to optimize front-end and back-end parameters, particularly in challenging low-light and HDR scenarios.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Motion-compensated EVIO framework</title>
<p>This section details the motion-compensated event image state estimation framework, which serves as the backbone of the methodology presented in <xref ref-type="sec" rid="s4">Section 4</xref>. Optimization of the state estimation parameters is addressed in the following paragraphs.</p>
<p>The motion-compensated EVIO system detects features within the edge image created from motion-compensated events by employing conventional image-based feature detection techniques. For example (<xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>), integrates event data with IMU data to obtain an accurate motion-compensated EVIO pipeline that leverages the distinctive features of event cameras to enable accurate state estimation in challenging scenarios. This approach is further extended in Ultimate SLAM (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>), where standard frames are incorporated as an additional sensing modality, achieving a more reliable and precise state estimation.</p>
<p>The motion-compensated EVIO system is traditionally divided into two parts: the front-end process, which processes a stream of events to establish feature tracks and triangulate landmarks, and the back-end, which integrates these feature tracks, landmarks, and IMU measurements to constantly update both current and past sensor states (<xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>). However, employing the edge image in this state estimation framework presents difficulties that often demand extensive parameter tuning.</p>
<p>To address these limitations, this work aims to enhance existing methods by developing an automated parameter optimization pipeline that facilitates the tuning process and identifies optimal parameters across diverse scenarios. The following paragraphs discuss key parameters that can be optimized within both the front-end and back-end components.</p>
<sec id="s3-1">
<label>3.1</label>
<title>Front-end process</title>
<p>The main approach in the front-end is to generate event frames from spatiotemporal clusters of events, followed by applying feature detection and tracking techniques. This state estimation system builds on methodologies from (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>; <xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>), where features are detected and tracked within edge images derived from motion-compensated events, employing conventional image-based feature detection and tracking methods. Particularly, the FAST corner detector (<xref ref-type="bibr" rid="B28">Rosten and Drummond, 2006</xref>) and the Lucas-Kanade tracker (<xref ref-type="bibr" rid="B17">Lucas and Kanade, 1981</xref>) are utilized for these purposes. Additionally, features from standard frames are extracted and incorporated into the back-end optimization module, enhancing overall robustness and accuracy.</p>
<p>In noise-free scenarios, event frames can be represented as <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the pixel value <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> indicates the elapsed time, and <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> signifies the polarity ranging from {-1,&#x2b;1}. Additionally, the events <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are synchronized by aligning them with the spatio-temporal windows of events based on the timestamps of the conventional frames. For each conventional frame at time <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, a new spatiotemporal event window <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is defined as follows:<disp-formula id="e1">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>Here, <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the index of the first event with a timestamp <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the size of the window. Subsequently, each spatiotemporal event window undergoes a transformation into an artificial event frame <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> by applying motion compensation at its individual timestamp, as demonstrated in the next equation:<disp-formula id="e2">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x3f5;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> represents the Kronecker delta, <inline-formula id="inf16">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the adjusted event location acquired by shifting event <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to align with the specified event camera frame. Further, it is necessary to adjust the movement of every event locally based on its respective timestamp due to the limited information in small window sizes and the motion blur introduced by extensive window sizes. The <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> in <xref ref-type="disp-formula" rid="e2">Equation 2</xref> can be calculated using the formula given by (<xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>), as shown in <xref ref-type="disp-formula" rid="e3">Equation 3</xref>:<disp-formula id="e3">
<mml:math id="m21">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf19">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the event pixel location, <inline-formula id="inf20">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>:</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the event camera projection sample derived from previous inherent calibration, <inline-formula id="inf21">
<mml:math id="m24">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> signifies the gradual transition of the camera poses at times <inline-formula id="inf22">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf23">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, derived from integrating the inertial measurements, and <inline-formula id="inf24">
<mml:math id="m27">
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the scene depth at time <inline-formula id="inf25">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> estimated through a 2D linear interpolation.</p>
<p>The count of events <inline-formula id="inf26">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in each spatiotemporal window needs to be adapted and can be optimized according to the texture density present in the scene. Hence, in this work, it has been chosen as one of the optimized parameters<xref ref-type="fn" rid="n1">
<sup>1</sup>
</xref>. The median depth of the current landmarks <inline-formula id="inf27">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can produce satisfactory results with reduced computational costs compared to linearly interpolating the depth <inline-formula id="inf28">
<mml:math id="m31">
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Therefore, the median depth of landmarks is optimized using the proposed method<xref ref-type="fn" rid="n1">
<sup>1</sup>
</xref> presented in <xref ref-type="sec" rid="s4">Section 4</xref>.</p>
<p>New features are identified using the FAST corner detector, which is applied to both motion-compensated event frames and standard frames (<xref ref-type="bibr" rid="B21">Mueggler et al., 2017a</xref>). This approach ensures an even distribution of features across the image by using a bucketing grid:<disp-formula id="e4">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mtext>if </mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>S</mml:mi>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mtext>if </mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>B</mml:mi>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mtext>if </mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf29">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the intensity at pixel <inline-formula id="inf30">
<mml:math id="m34">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf31">
<mml:math id="m35">
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the threshold value, <inline-formula id="inf32">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the intensity difference between pixel <inline-formula id="inf33">
<mml:math id="m37">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf34">
<mml:math id="m38">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf35">
<mml:math id="m39">
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the darker corner, <inline-formula id="inf36">
<mml:math id="m40">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the similar pixel, and <inline-formula id="inf37">
<mml:math id="m41">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the brighter corner. To effectively detect these features, the threshold of the FAST detector is optimized in this work<xref ref-type="fn" rid="n1">
<sup>1</sup>
</xref>. These features are subsequently tracked from <inline-formula id="inf38">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf39">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, derived through an incremental transformation. Moreover, landmarks are tracked using the pyramidal Lukas-Kanade tracking algorithm (<xref ref-type="bibr" rid="B17">Lucas and Kanade, 1981</xref>), with the number of pyramid levels for feature extraction set as an automatically adjusted parameter<xref ref-type="fn" rid="n1">
<sup>1</sup>
</xref>. Furthermore, a two-point RANSAC approach (<xref ref-type="bibr" rid="B20">Mueggler et al., 2014</xref>) is used for additional filtering of outlier feature tracks. In this system, the parameters for detection and tracking are maintained consistently across both motion-compensated event frames and conventional frames. Moreover, if the number of tracked features drops below a certain threshold <inline-formula id="inf40">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, features are re-detected.</p>
</sec>
<sec id="s3-2">
<label>3.2</label>
<title>Back-end process</title>
<p>This section explores the integration of feature tracks from the event stream with IMU data, using a smoothing-based nonlinear optimization method on chosen keyframes. A detailed comprehensive analysis of IMU biases and kinematics can be found in (<xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>; <xref ref-type="bibr" rid="B10">Guan et al., 2023</xref>). The visual-inertial nonlinear optimization is described by a cost function <inline-formula id="inf41">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>J</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">V IO</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which consists of three components: two weighted reprojection errors associated with event-based and conventional camera data, and an inertial error <inline-formula id="inf42">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The cost function <inline-formula id="inf43">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>J</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">V IO</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is formulated as shown in <xref ref-type="disp-formula" rid="e5">Equation 5</xref>:<disp-formula id="e5">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>J</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">V IO</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>&#x3f5;</mml:mi>
<mml:mi>&#x237;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>and the reprojection error <inline-formula id="inf44">
<mml:math id="m49">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is given by <xref ref-type="disp-formula" rid="e6">Equation 6</xref>:<disp-formula id="e6">
<mml:math id="m50">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>In the previous equations, <inline-formula id="inf45">
<mml:math id="m51">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the sensor identifier, <inline-formula id="inf46">
<mml:math id="m52">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> signifies the frame identifier, and <inline-formula id="inf47">
<mml:math id="m53">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> refers to the landmark. The set <inline-formula id="inf48">
<mml:math id="m54">
<mml:mrow>
<mml:mi>&#x237;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> includes the landmarks tracked by sensor <inline-formula id="inf49">
<mml:math id="m55">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in the <inline-formula id="inf50">
<mml:math id="m56">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> frame. The data matrix for each landmark measurement <inline-formula id="inf51">
<mml:math id="m57">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is represented as <inline-formula id="inf52">
<mml:math id="m58">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, while <inline-formula id="inf53">
<mml:math id="m59">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the data matrix corresponding to the IMU error in the <inline-formula id="inf54">
<mml:math id="m60">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> frame. Additionally, <inline-formula id="inf55">
<mml:math id="m61">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the calculated image coordinates for every <inline-formula id="inf56">
<mml:math id="m62">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> frame. The IMU error is computed as the disparity between predicted and actual trajectories (<xref ref-type="bibr" rid="B15">Leutenegger et al., 2013</xref>). Optimization is performed selectively, focusing on a subset comprising of keyframes and the last <inline-formula id="inf57">
<mml:math id="m63">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> frames in a sliding window, while predictions for intervening frames are propagated using IMU data. The number of keyframes employed in the back-end process is one of the parameters optimized in this study<xref ref-type="fn" rid="n1">
<sup>1</sup>
</xref>.</p>
</sec>
</sec>
<sec sec-type="materials|methods" id="s4">
<label>4</label>
<title>Materials and methods</title>
<p>Motion compensation necessitates identifying motion parameters that precisely match a sequence of events. By using a continuous-time warping framework, it is possible to fully leverage the exact temporal information offered by events, setting this approach apart from conventional image-based methods. Obtaining parameters for these continuous-time motion models frequently relies on optimization strategies (<xref ref-type="bibr" rid="B23">Mueggler et al., 2018</xref>). This section focuses on optimizing Event SLAM parameters to enhance trajectory pose prediction using the BGD algorithm. A comprehensive diagram of the proposed VIO-GO process flow is depicted in <xref ref-type="fig" rid="F2">Figure 2</xref>. Therefore, BGD is chosen in this study for its proven stability and reliability in achieving efficient parameter tuning.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>A detailed illustration showcasing the VIO-GO process, highlighting the integration of event frames, standard frames, and IMU using BGD optimization. The figure provides an overview of the iterative optimization process, emphasizing the seamless fusion of event-based visual information and inertial measurements to refine the estimated trajectory and reduce the MPE.</p>
</caption>
<graphic xlink:href="frobt-12-1541017-g002.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a Visual-Inertial Odometry (VIO) system, divided into three sections: VIO Front End, VIO Back End, and Proposed VIO-GO Block. The VIO Front End processes event and standard frames for feature extraction and triangulation. The VIO Back End uses IMU data for pose estimation and nonlinear optimization, comparing with ground truth. The Proposed VIO-GO Block refines parameters using gradient descent and mean position error across different VIO modules.</alt-text>
</graphic>
</fig>
<p>BGD is chosen due to its fundamental role as an optimization technique widely used in machine learning (<xref ref-type="bibr" rid="B24">Mustapha et al., 2020</xref>). Its significance emerges from its ability to systematically uncover optimal parameter values through iterative adjustments guided by gradients of the objective function, computed across the entire dataset. This makes BGD particularly effective for smaller datasets, such as ours, where the dataset is dynamically generated as the robot navigates through the environment. Unlike SGD, which updates parameters based on individual data points and can introduce noise, BGD provides stable convergence, minimizing variance and ensuring more consistent results (<xref ref-type="bibr" rid="B30">Singh and Singh, 2023</xref>). Furthermore, successful applications of BGD in parameter optimization, such as in (<xref ref-type="bibr" rid="B24">Mustapha et al., 2020</xref>; <xref ref-type="bibr" rid="B25">Rao et al., 2023</xref>), demonstrate its robustness and effectiveness in enhancing model performance. Therefore, BGD is chosen in this study for its demonstrated stability and reliability in achieving efficient parameter tuning.</p>
<p>A key challenge with the GD method is that the search may oscillate within the search space, influenced by the gradient&#x2019;s direction. For instance, although the descent can move toward a global minimum, it may sometimes veer off course due to local minima or saddle points, ultimately slowing convergence. To address this, a common solution is to introduce momentum into the parameter update equation. This approach introduces an additional hyperparameter that controls the extent to which the past gradient (momentum) influences the current update (<xref ref-type="bibr" rid="B5">Chandra et al., 2022</xref>). Momentum helps the search maintain a consistent direction, reducing oscillations and enhancing the likelihood of bypassing local minima. In this work, momentum has been added to the BGD algorithm, formulated as shown in <xref ref-type="disp-formula" rid="e7">Equation 7</xref>:<disp-formula id="e7">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf58">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> defines the adjusted gradient incorporating momentum, <inline-formula id="inf59">
<mml:math id="m66">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the hyperparameter that represents the momentum constant, and <inline-formula id="inf60">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the gradient, showing the direction of decrease for the cost function.</p>
<p>Identifying the optimal event window size is crucial for event-based SLAM systems that use motion compensation to represent event data. This calibration relies on the event frame&#x2019;s dynamics, influenced more by camera resolution and scene complexity than by the speed of camera motion (<xref ref-type="bibr" rid="B34">Xiao et al., 2022</xref>). The number of events <inline-formula id="inf61">
<mml:math id="m68">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in every spatiotemporal window must be adjusted based on the scene&#x2019;s texture density, making it a main optimization target in VIO-GO. The primary goal is to achieve sharp motion blur-free edges, ensuring that the event frame accurately reflects the scene&#x2019;s layout.</p>
<p>The VIO-GO model is implemented alongside the state-of-the-art VIO method, Ultimate SLAM, chosen for its use of motion-compensated images to represent event data, which requires significant parameter adjustments. VIO-GO functions as an auxiliary technique that automatically finds and updates optimal parameters within the Ultimate SLAM framework.</p>
<p>VIO-GO incorporates several approaches: the 2-parameter set (VIO-GO2), the 4-parameter set (VIO-GO4), the 6-parameter set (VIO-GO6), and the 8-parameter set (VIO-GO8). The parameters selected for each approach are determined from the front-end and back-end equations discussed in <xref ref-type="sec" rid="s3">Section 3</xref>. VIO-GO2 and VIO-GO4 concentrate on optimizing the spatiotemporal event window parameters, while VIO-GO6 and VIO-GO8 extend optimization to include feature extraction and back-end parameters. The key parameter sets <inline-formula id="inf62">
<mml:math id="m69">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> considered for optimizing event VIO are detailed in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>List of parameter sets <inline-formula id="inf63">
<mml:math id="m70">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> selected for BGD optimization in event-based VIO.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="3" align="center">VIO-GO2</th>
</tr>
<tr>
<th align="center">Parameter</th>
<th align="center">Symbol</th>
<th align="center">Explanation</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Frame size</td>
<td align="center">
<inline-formula id="inf64">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Number of events drawn from the event camera</td>
</tr>
<tr>
<td align="left">Noise event rate</td>
<td align="center">
<inline-formula id="inf65">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Events per second regarded as noise</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th colspan="3" align="center">VIO-GO4</th>
</tr>
<tr>
<th align="center">Parameter</th>
<th align="center">Symbol</th>
<th align="center">Explanation</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Frame size</td>
<td align="center">
<inline-formula id="inf66">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Number of events drawn from the event camera</td>
</tr>
<tr>
<td align="left">Noise event rate</td>
<td align="center">
<inline-formula id="inf67">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Events per second regarded as noise</td>
</tr>
<tr>
<td align="left">Data size augmented event packet</td>
<td align="center">
<inline-formula id="inf68">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Event packet size</td>
</tr>
<tr>
<td align="left">Frame norm factor</td>
<td align="center">
<inline-formula id="inf69">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Normalization factor for event frames</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th colspan="3" align="center">VIO-GO6</th>
</tr>
<tr>
<th align="center">Parameter</th>
<th align="center">Symbol</th>
<th align="center">Description</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Frame size</td>
<td align="center">
<inline-formula id="inf70">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Number of events drawn from the event camera</td>
</tr>
<tr>
<td align="left">Noise event rate</td>
<td align="center">
<inline-formula id="inf71">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Events per second regarded as noise</td>
</tr>
<tr>
<td align="left">VIO median depth</td>
<td align="center">
<inline-formula id="inf72">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Median depth of landmarks</td>
</tr>
<tr>
<td align="left">Imp detector num octaves</td>
<td align="center">
<inline-formula id="inf73">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Number of pyramid levels for feature extraction</td>
</tr>
<tr>
<td align="left">Imp detector threshold</td>
<td align="center">
<inline-formula id="inf74">
<mml:math id="m81">
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Absolute threshold value of the FAST detector</td>
</tr>
<tr>
<td align="left">VIO numkeyframes</td>
<td align="center">
<inline-formula id="inf75">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Number of keyframes in back-end process</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th colspan="3" align="center">VIO-GO8</th>
</tr>
<tr>
<th align="center">Parameter</th>
<th align="center">Symbol</th>
<th align="center">Description</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Frame size</td>
<td align="center">
<inline-formula id="inf76">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Number of events drawn from the event camera</td>
</tr>
<tr>
<td align="left">Noise event rate</td>
<td align="center">
<inline-formula id="inf77">
<mml:math id="m84">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Events per second regarded as noise</td>
</tr>
<tr>
<td align="left">VIO median depth</td>
<td align="center">
<inline-formula id="inf78">
<mml:math id="m85">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Median depth of landmarks</td>
</tr>
<tr>
<td align="left">Imp detector num octaves</td>
<td align="center">
<inline-formula id="inf79">
<mml:math id="m86">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Number of pyramid levels for feature extraction</td>
</tr>
<tr>
<td align="left">Imp detector threshold</td>
<td align="center">
<inline-formula id="inf80">
<mml:math id="m87">
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Absolute threshold value of the FAST detector</td>
</tr>
<tr>
<td align="left">VIO numkeyframes</td>
<td align="center">
<inline-formula id="inf81">
<mml:math id="m88">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Number of keyframes in the back-end process</td>
</tr>
<tr>
<td align="left">VIO kfselect numfts lower thresh</td>
<td align="center">
<inline-formula id="inf82">
<mml:math id="m89">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Force keyframe selection below this number of features</td>
</tr>
<tr>
<td align="left">Detector max features per frame</td>
<td align="center">
<inline-formula id="inf83">
<mml:math id="m90">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Maximum number of features to extract per frame</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>All VIO-GO approaches prioritize the event window size from <xref ref-type="disp-formula" rid="e1">Equation 1</xref> as the main optimization parameter, due to its critical role, as discussed previously. Furthermore, each method adjusts the noise event rate, which acts as a threshold for scenarios where the sensor is stationary and produces minimal events. When the event rate falls below this threshold, indicating low activity aside from noise events, the sensor is held in a stationary state. These two parameters are the focus of fine-tuning in VIO-GO2.</p>
<p>The optimal values of <inline-formula id="inf84">
<mml:math id="m92">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for the VIO-GO2 are calculated from the set of BGD equations, as shown in <xref ref-type="disp-formula" rid="e8">Equation 8</xref>:<disp-formula id="e8">
<mml:math id="m93">
<mml:mrow>
<mml:mtable class="align" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf85">
<mml:math id="m94">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the gradient frame size and <inline-formula id="inf86">
<mml:math id="m95">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> indicates the gradient noise event rate.</p>
<p>In VIO-GO4, additional parameters are optimized, including the event packet size, which defines the dimensions of augmented event packets sent to the front-end for rendering event frames, and the normalization factor for event frames. However, in VIO-GO6 and VIO-GO8, these parameters were adjusted, as they were found to have minimal impact on the estimated trajectory of the overall VIO system, as discussed in <xref ref-type="sec" rid="s5">Section 5</xref>.</p>
<p>The optimal values of <inline-formula id="inf87">
<mml:math id="m96">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for the VIO-GO4 are calculated from the set of BGD equations, as shown in <xref ref-type="disp-formula" rid="e9">Equation 9</xref>: <disp-formula id="e9"> <mml:math id="m91"> <mml:mrow> <mml:mtable class="align" columnalign="left"> <mml:mtr> <mml:mtd columnalign="right"/> <mml:mtd columnalign="left"> <mml:msub> <mml:mrow> <mml:mi>F</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>S</mml:mi> </mml:mrow> </mml:msub> <mml:mo>&#x3d;</mml:mo> <mml:msub> <mml:mrow> <mml:mi>F</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>S</mml:mi> </mml:mrow> </mml:msub> <mml:mo>&#x2212;</mml:mo> <mml:mfenced open="(" close=")"> <mml:mrow> <mml:mi>&#x3b3;</mml:mi> <mml:mo>&#x2217;</mml:mo> <mml:msub> <mml:mrow> <mml:mi>G</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>F</mml:mi> <mml:mi>S</mml:mi> </mml:mrow> </mml:msub> </mml:mrow> </mml:mfenced> </mml:mtd> </mml:mtr> <mml:mtr> <mml:mtd columnalign="right"/> <mml:mtd columnalign="left"> <mml:msub> <mml:mrow> <mml:mi>N</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>E</mml:mi> <mml:mi>R</mml:mi> </mml:mrow> </mml:msub> <mml:mo>&#x3d;</mml:mo> <mml:msub> <mml:mrow> <mml:mi>N</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>E</mml:mi> <mml:mi>R</mml:mi> </mml:mrow> </mml:msub> <mml:mo>&#x2212;</mml:mo> <mml:mfenced open="(" close=")"> <mml:mrow> <mml:mi>&#x3b3;</mml:mi> <mml:mo>&#x2217;</mml:mo> <mml:mfrac> <mml:mrow> <mml:msub> <mml:mrow> <mml:mi>G</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>N</mml:mi> </mml:mrow> </mml:msub> </mml:mrow> <mml:mrow> <mml:mn>2</mml:mn> </mml:mrow> </mml:mfrac> </mml:mrow> </mml:mfenced> </mml:mtd> </mml:mtr> <mml:mtr> <mml:mtd columnalign="right"/> <mml:mtd columnalign="left"> <mml:msub> <mml:mrow> <mml:mi>E</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>P</mml:mi> </mml:mrow> </mml:msub> <mml:mo>&#x3d;</mml:mo> <mml:msub> <mml:mrow> <mml:mi>E</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>P</mml:mi> </mml:mrow> </mml:msub> <mml:mo>&#x2212;</mml:mo> <mml:mfenced open="(" close=")"> <mml:mrow> <mml:mi>&#x3b3;</mml:mi> <mml:mo>&#x2217;</mml:mo> <mml:mn>1.5</mml:mn> <mml:mo>&#x2217;</mml:mo> <mml:msub> <mml:mrow> <mml:mi>G</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>E</mml:mi> <mml:mi>P</mml:mi> </mml:mrow> </mml:msub> </mml:mrow> </mml:mfenced> </mml:mtd> </mml:mtr> <mml:mtr> <mml:mtd columnalign="right"/> <mml:mtd columnalign="left"> <mml:msub> <mml:mrow> <mml:mi>N</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>F</mml:mi> </mml:mrow> </mml:msub> <mml:mo>&#x3d;</mml:mo> <mml:msub> <mml:mrow> <mml:mi>G</mml:mi> </mml:mrow> <mml:mrow> <mml:mi>N</mml:mi> <mml:mi>F</mml:mi> </mml:mrow> </mml:msub> <mml:mo>,</mml:mo> </mml:mtd> </mml:mtr> </mml:mtable> </mml:mrow> </mml:math>  <label>(9)</label>
</disp-formula> </p> <p>where <inline-formula id="inf88">
<mml:math id="m97">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the gradient frame size, <inline-formula id="inf89">
<mml:math id="m98">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> corresponds to the gradient noise event rate, <inline-formula id="inf90">
<mml:math id="m99">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> corresponds to the gradient event packet size, and <inline-formula id="inf91">
<mml:math id="m100">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the gradient normalization factor.</p>
<p>To enhance feature identification in <xref ref-type="disp-formula" rid="e4">Equation 4</xref>, both VIO-GO6 and VIO-GO8 adjust the FAST detector threshold and the number of pyramid levels used for feature extraction. Additionally, both methods fine-tune the parameter defining the number of keyframes used in the back-end optimization process.</p>
<p>The optimal values of <inline-formula id="inf92">
<mml:math id="m101">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for the VIO-GO6 are calculated from the set of BGD equations, as shown in <xref ref-type="disp-formula" rid="e10">Equation 10</xref>:<disp-formula id="e10">
<mml:math id="m102">
<mml:mrow>
<mml:mtable class="align" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mi>T</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">CKF</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>where <inline-formula id="inf93">
<mml:math id="m103">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf94">
<mml:math id="m104">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> remain as in the previous approach, <inline-formula id="inf95">
<mml:math id="m105">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> corresponds to the gradient of the median depth, <inline-formula id="inf96">
<mml:math id="m106">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the gradient of the number of octaves, <inline-formula id="inf97">
<mml:math id="m107">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the gradient threshold, and <inline-formula id="inf98">
<mml:math id="m108">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">CKF</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> corresponds to the gradient of the number of keyframes in the back-end.</p>
<p>VIO-GO8 includes two additional parameters not found in VIO-GO6: the minimum number of features needed to enforce keyframe selection and the maximum number of features to extract from each frame. These parameters have a considerable effect on the feature extraction process, thereby affecting the overall performance of the VIO system.</p>
<p>The optimal values of <inline-formula id="inf99">
<mml:math id="m109">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for VIO-GO8 are calculated from the set of BGD equations, as shown in <xref ref-type="disp-formula" rid="e11">Equation 11</xref>:</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>Proposed parameter set <inline-formula id="inf100">
<mml:math id="m110">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> of optimization method<xref ref-type="fn" rid="n1">
<sup>1</sup>
</xref>.<list list-type="simple">
<list-item>
<p>
<bold>Input:Initial parameters</bold> <inline-formula id="inf101">
<mml:math id="m111">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>, Number of iterations</bold> <inline-formula id="inf102">
<mml:math id="m112">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>, Learning rate</bold> <inline-formula id="inf103">
<mml:math id="m113">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<bold>Output: Final parameters</bold> <inline-formula id="inf104">
<mml:math id="m114">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>1.&#x2003;<bold>for</bold> <inline-formula id="inf105">
<mml:math id="m115">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> <bold>to</bold> <inline-formula id="inf106">
<mml:math id="m116">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>2.&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;estimate <inline-formula id="inf107">
<mml:math id="m117">
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>U</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>3.&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;compute <inline-formula id="inf108">
<mml:math id="m118">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x2207;</mml:mi>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>4.&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf109">
<mml:math id="m119">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2254;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>5.&#x2003;<bold>return</bold> <inline-formula id="inf110">
<mml:math id="m120">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
</p>
</statement>
<disp-formula id="e11">
<mml:math id="m121">
<mml:mrow>
<mml:mtable class="align" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mi>T</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">CKF</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">DMF</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf111">
<mml:math id="m122">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the gradient of the maximum number of features per frame and <inline-formula id="inf112">
<mml:math id="m123">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">DMF</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the gradient keyframes selection threshold, while all other parameters remain the same as in VIO-GO6.</p>
<p>The loss functions for these parameter sets are calculated based on the mean error and target error of the trajectories obtained from the event-based VIO. The parameters are updated using a learning rate <inline-formula id="inf113">
<mml:math id="m124">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of 0.02. Once these parameters are optimized, they are fedback into the event-based VIO, and the resulting trajectories in the <inline-formula id="inf114">
<mml:math id="m125">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf115">
<mml:math id="m126">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf116">
<mml:math id="m127">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> directions are recorded along with the MPE. The error is calculated over a 5-s interval, as described in (<xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>; <xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>). The complete parameter optimization technique is outlined in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>.</p>
<p>Although the Ultimate SLAM involves numerous parameters, restricting their number ensures the practical feasibility of the proposed system. This constraint was chosen for two main reasons. First, the selected parameters are crucial elements of the front-end and back-end equations discussed in <xref ref-type="sec" rid="s3">Section 3</xref>. Second, maintaining a fixed learning rate across all VIO-GO approaches makes it challenging to integrate parameters with significantly different values into the GD equations. Moreover, using various learning rates for different parameters would significantly increase the system&#x2019;s computational cost.</p>
<p>The proposed algorithm extends its applicability beyond Ultimate SLAM, demonstrating adaptability to a broader range of algorithms. Specifically, the VIO-GO2 and VIO-GO4 approaches are applicable to any event-based VIO system that uses motion-compensated images for event data representation. This compatibility is due to the shared use of a spatio-temporal event window in the front-end processing of these systems. Moreover, VIO-GO6 and VIO-GO8 are designed for seamless integration with event-based systems that specifically use the FAST detector for feature extraction and nonlinear back-end optimization.</p>
</sec>
<sec sec-type="results" id="s5">
<label>5</label>
<title>Results</title>
<p>We assess the efficiency of the VIO-GO framework by comparing it to various event-based VIO methods across challenging sequences from the Event Camera Dataset (<xref ref-type="bibr" rid="B22">Mueggler et al., 2017b</xref>). This dataset comprises sequences captured with a Dynamic and Active-pixel Vision Sensor (DAVIS) across various synthetic and real-world environments, serving as a widely accepted benchmark for evaluating SLAM systems for high-speed motion and HDR scenarios. The sequences exhibit complexity through varying speeds, scenes, and DoF. In the shapes, poster, and boxes datasets, each DoF is initially excited individually, followed by mixed and progressively faster excitations, resulting in higher event rates over time. The HDR datasets include significant intrascene contrasts created by a spotlight. The dynamic sequences, gathered in a simulated office environment and observed by a motion-capture method, depict an individual transitioning from sitting at a desk to moving around. <xref ref-type="fig" rid="F3">Figure 3</xref> displays snapshots from representative sequences within the dataset, highlighting the diversity and complexity of the captured scenarios.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Scenes from the different sequences in the Event Camera Dataset (<xref ref-type="bibr" rid="B22">Mueggler et al., 2017b</xref>, Copyright &#xa9; 2017 by The Author(s). Reprinted by Permission of Sage Publications). <bold>(a)</bold> Boxes_sequence. <bold>(b)</bold> Dynamic_sequence. <bold>(c)</bold> Shapes_sequence.</p>
</caption>
<graphic xlink:href="frobt-12-1541017-g003.tif">
<alt-text content-type="machine-generated">(a) Several large, decorative cubes with various textures displayed on a colorful patterned floor. (b) A person seated at a table with books and a drone, focusing on reading. (c) Wall with black geometric shapes on paper, including circles and stars, arranged in a cluster.</alt-text>
</graphic>
</fig>
<p>Our evaluation includes a quantitative examination to assess the accuracy of the proposed algorithm. Accuracy is measured using the MPE, expressed as a percentage of the total distance traveled. A 6-DOF transformation in SE(3) is applied over a 5-s segment of the trajectory to align the estimated and ground truth trajectories. This alignment and accuracy calculation is carried out using the EVO tool (<xref ref-type="bibr" rid="B9">Grupp, 2017</xref>). All experiments were conducted on a laptop powered by an Apple M1 chip, running Ubuntu 20.04 and ROS Noetic. To evaluate the performance of the presented adaptive optimization system, it was integrated with the Ultimate SLAM framework (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>). Ultimate SLAM uses edge images for VIO, requiring significant parameter tuning to adapt to the dynamic nature of events in the scene.</p>
<p>To initiate the BGD optimization process, we set all parameter values to the upper bounds of their respective ranges. This choice provides a conservative starting point, allowing the system to iteratively refine the parameters toward their optimal values. For IMU biases, fixed initial values were used throughout all experiments to ensure consistent benchmarking. These values are derived from the calibration data provided with the Event Camera Dataset <xref ref-type="bibr" rid="B22">Mueggler et al., 2017b</xref>. Moreover, they fall within the nominal factory calibration ranges specified in the datasheet of the InvenSense MPU-6150 IMU sensor<xref ref-type="fn" rid="n2">
<sup>2</sup>
</xref>, which is the integrated IMU sensor in DAVIS. The values used are listed in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Initial IMU bias values used in all experiments.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Accelerometer</th>
<th align="center">Value (m/s<sup>2</sup>)</th>
<th align="center">Gyroscope</th>
<th align="center">Value (rad/s)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Bias X</td>
<td align="center">
<inline-formula id="inf117">
<mml:math id="m128">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>0.1059</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">Bias X</td>
<td align="center">0.0494</td>
</tr>
<tr>
<td align="center">Bias Y</td>
<td align="center">
<inline-formula id="inf118">
<mml:math id="m129">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>0.2015</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">Bias Y</td>
<td align="center">0.0105</td>
</tr>
<tr>
<td align="center">Bias Z</td>
<td align="center">0.2432</td>
<td align="center">Bias Z</td>
<td align="center">0.0012</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The evaluation is divided into two parts. First, we analyze various VIO-GO approaches to identify the most effective model, determine the optimal number of parameters for optimization, and test the scalability of the presented approach. Next, in the second part, we compare VIO-GO with the state-of-the-art edge image-based event-driven VIO approaches to highlight its performance advantages.</p>
<sec id="s5-1">
<label>5.1</label>
<title>Evaluating VIO-GO approaches</title>
<p>We evaluate the impact of varying the number of optimized parameters in the VIO-GO approach on the overall performance of the VIO system. This involves comparing VIO-GO2, VIO-GO4, VIO-GO6, and VIO-GO8 across various sequences from the Event Camera Dataset. <xref ref-type="table" rid="T3">Table 3</xref> provides a detailed comparison of the results.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The performance of various VIO-GO approaches measured in terms of MPE (%).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Dataset</th>
<th align="center">VIO-GO2 (2 parameters)</th>
<th align="center">VIO-GO4 (4 parameters)</th>
<th align="center">VIO-GO6 (6 parameters)</th>
<th align="center">VIO-GO8 (8 parameters)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">boxes_6dof</td>
<td align="center">0.45</td>
<td align="center">0.50</td>
<td align="center">0.44</td>
<td align="center">
<bold>0.41</bold>
</td>
</tr>
<tr>
<td align="left">boxes_translation</td>
<td align="center">0.35</td>
<td align="center">0.27</td>
<td align="center">0.26</td>
<td align="center">
<bold>0.25</bold>
</td>
</tr>
<tr>
<td align="left">dynamic_6dof</td>
<td align="center">0.29</td>
<td align="center">0.35</td>
<td align="center">
<bold>0.27</bold>
</td>
<td align="center">
<bold>0.27</bold>
</td>
</tr>
<tr>
<td align="left">dynamic_translation</td>
<td align="center">0.33</td>
<td align="center">0.27</td>
<td align="center">0.26</td>
<td align="center">
<bold>0.25</bold>
</td>
</tr>
<tr>
<td align="left">hdr_boxes</td>
<td align="center">0.48</td>
<td align="center">0.46</td>
<td align="center">0.37</td>
<td align="center">
<bold>0.35</bold>
</td>
</tr>
<tr>
<td align="left">hdr_poster</td>
<td align="center">0.29</td>
<td align="center">0.31</td>
<td align="center">0.31</td>
<td align="center">
<bold>0.25</bold>
</td>
</tr>
<tr>
<td align="left">poster_6dof</td>
<td align="center">0.59</td>
<td align="center">0.69</td>
<td align="center">0.54</td>
<td align="center">
<bold>0.50</bold>
</td>
</tr>
<tr>
<td align="left">poster_translation</td>
<td align="center">0.26</td>
<td align="center">0.26</td>
<td align="center">0.25</td>
<td align="center">
<bold>0.23</bold>
</td>
</tr>
<tr>
<td align="left">shapes_6dof</td>
<td align="center">1.05</td>
<td align="center">0.91</td>
<td align="center">
<bold>0.77</bold>
</td>
<td align="center">
<bold>0.77</bold>
</td>
</tr>
<tr>
<td align="left">shapes_translation</td>
<td align="center">0.64</td>
<td align="center">0.50</td>
<td align="center">
<bold>0.33</bold>
</td>
<td align="center">0.36</td>
</tr>
<tr>
<td align="left">Average</td>
<td align="center">0.47</td>
<td align="center">0.45</td>
<td align="center">0.38</td>
<td align="center">
<bold>0.36</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The values displayed in bold show the best results.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The results show that increasing the number of tuned parameters in the proposed model significantly enhances the overall performance of the VIO system. As illustrated in <xref ref-type="table" rid="T3">Table 3</xref>, optimizing 8 parameters (VIO-GO8) instead of 2 (VIO-GO2) results in a 24% reduction in the average MPE of the estimated trajectory across all sequences. Similarly, VIO-GO4 surpasses VIO-GO2 in most sequences, achieving a 4% reduction in average MPE. Further improvements are observed with VIO-GO6, which reduces the average MPE by 16% compared to VIO-GO4. Finally, VIO-GO8 delivers the most accurate trajectories, achieving an additional 5% reduction in average MPE compared to VIO-GO6 across all tested sequences.</p>
<p>
<xref ref-type="fig" rid="F4">Figure 4</xref> presents heatmaps of the estimated trajectories obtained from various VIO-GO approaches for the <italic>hdr_boxes</italic> sequence, aligned with the ground truth trajectory. The plots demonstrate that all VIO-GO variants produce precise trajectory estimations, as indicated by the low APE values. Notably, the graphs highlight clear improvement in trajectory accuracy with an increasing number of optimized parameters. This is shown by the significant reduction in the mean APE from 0.031 m for VIO-GO2 to 0.020 m for VIO-GO8. <xref ref-type="fig" rid="F5">Figure 5</xref> presents relative error metrics to evaluate the performance of different VIO-GO approaches on the <italic>hdr_boxes</italic> and <italic>boxes_translation</italic> sequences. The charts clearly demonstrate the effectiveness of VIO-GO in reducing trajectory drift over time, with notable improvements observed as the number of optimized parameters increases.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Heatmaps depicting the APE of various VIO-GO trajectories for the <italic>hdr_boxes</italic> sequence, aligned with the ground truth via a 6-DOF transformation over a 5-s period using the EVO tool. <bold>(a)</bold> VIO-GO2. <bold>(b)</bold> VIO-GO4. <bold>(c)</bold> VIO-GO6. <bold>(d)</bold> VIO-GO8.</p>
</caption>
<graphic xlink:href="frobt-12-1541017-g004.tif">
<alt-text content-type="machine-generated">Four graphs labeled (a) VIO-G02, (b) VIO-G04, (c) VIO-G06, and (d) VIO-G08 display APE with respect to translation part (in meters) using SE(3) Umeyama alignment. Each graph has a colored trajectory with a reference line. The x-axis indicates x (in meters), and the y-axis shows z (in meters). A color gradient represents error values, transitioning from blue to red, with respective color bars on the right.</alt-text>
</graphic>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Comparison of relative errors across various VIO-GO variants. <bold>(a)</bold> <italic>Hdr_boxes</italic> sequence. <bold>(b)</bold> <italic>Boxes_translation</italic> sequence.</p>
</caption>
<graphic xlink:href="frobt-12-1541017-g005.tif">
<alt-text content-type="machine-generated">Two box plot graphs display translational error over time in meters. The top graph, labeled &#x22;Hdr_boxes sequence,&#x22; shows data for VIO-GO2, VIO-GO4, VIO-GO6, and VIO-GO8, with VIO-GO2 having the highest initial error. The bottom graph, labeled &#x22;Boxes_translation sequence,&#x22; presents similar data patterns but with lower error magnitudes. Both graphs range from zero to about fifty-five seconds and include legend markers for each data set.</alt-text>
</graphic>
</fig>
<p>To further evaluate computational efficiency, we measured the elapsed time of each VIO-GO configuration (with 2, 4, 6, and 8 parameters) using the same hardware setup. The experiments show that the VIO-GO8 consistently achieves lower elapsed time across most sequences, with an average elapsed time of 16.39 s, compared to 18.98 s for VIO-GO2, 18.51 s for VIO-GO4, and 18.90 s for VIO-GO6. Therefore, an average computational improvement of 13.6% over VIO-GO2 was observed, primarily due to the fine-tuning of key parameters. This computational efficiency gain is achieved during the feature extraction phase, specifically the number of features used to trigger keyframe selection and the maximum number of features extracted per frame. By optimizing these parameters, VIO-GO8 reduces the computational overhead associated with processing redundant or suboptimal features, leading to faster execution. It is important to note that VIO-GO8 requires a higher optimization cost upfront compared to other approaches, due to the increased number of parameters being tuned. Nevertheless, this does not significantly impact the time required to find the optimal parameters, as the BGD algorithm efficiently explores the parameter space in parallel and requires minimal computational resources.</p>
<p>Although VIO-GO8 delivers the best results in terms of both accuracy and computational efficiency, it is worth noting that increasing the number of optimized parameters beyond eight may further enhance performance. However, such an expansion would also introduce greater complexity into the optimization process. As noted previously, the choice to limit the parameter set to eight was driven by practical considerations, including the constraints of maintaining a fixed learning rate and controlling computational overhead associated with parameter tuning. Nonetheless, extending the optimization to a broader set of parameters remains a promising direction for future research.</p>
</sec>
<sec id="s5-2">
<label>5.2</label>
<title>Comparing with event-based VIO methods</title>
<p>In our evaluation, we benchmark the proposed system against the raw results of Ultimate SLAM, as reported by its authors who used per-sequence parameter tuning and accurate IMU bias initialization. Building upon the Ultimate SLAM framework, our model is evaluated against this baseline to demonstrate its capability to automatically identify optimal parameters for each sequence in the Event Camera Dataset. Additionally, we compare it with Ultimate SLAM results obtained using a fixed parameter set adjusted across all sequences simultaneously and initialized with zero IMU bias, as presented in (<xref ref-type="bibr" rid="B19">Mahlknecht et al., 2022</xref>).</p>
<p>The aim of this comparison is to demonstrate the importance of parameter optimization in event-based VIO methods and to highlight the performance of the proposed model against a fixed parameter set across various scenarios. Moreover, we compare VIO-GO with (<xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>), an event-based algorithm coupled with an IMU, considered the foundational pipeline for Ultimate SLAM. The evaluation also includes EKLT-VIO (<xref ref-type="bibr" rid="B19">Mahlknecht et al., 2022</xref>), a system that integrates the EKLT feature tracker with a filter-based back-end, and EVIO (<xref ref-type="bibr" rid="B35">Zhu et al., 2017</xref>), an event-based tracking algorithm combined with an IMU. Similar to the proposed approach, both EKLT-VIO and EVIO are developed to perform efficiently under diverse conditions, including HDR environments and different lighting scenarios. The developers of the selected EVIO methods evaluated them using MPE as the error metric and the Event Camera Dataset as the simulation environment, following the same evaluation methodology employed in this work.</p>
<p>
<xref ref-type="table" rid="T4">Table 4</xref> provides a comprehensive comparison of the MPE five benchmark algorithms and VIO-GO using the 8-parameter configuration (VIO-GO8), across various sequences from the Event Camera Dataset. As shown in <xref ref-type="table" rid="T4">Table 4</xref>, the presented integrated system achieves state-of-the-art performance. Compared to Ultimate SLAM with a fixed parameter set (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>; <xref ref-type="bibr" rid="B19">Mahlknecht et al., 2022</xref>), which has an average MPE of 0.89%, VIO-GO8 demonstrates superior performance across all sequences with an average MPE of 0.36%. In contrast to the raw results of Ultimate SLAM (<xref ref-type="bibr" rid="B33">Vidal et al., 2018</xref>), VIO-GO8 successfully identifies optimal parameters, resulting in a lower MPE in the <italic>boxes_translation</italic>, <italic>hdr_boxes</italic>, and <italic>hdr_poster</italic> sequences, with MPE values of 0.25%, 0.35%, and 0.25%, respectively. Although the raw results of Ultimate SLAM exhibit better performance compared to VIO-GO, it is important to note that Ultimate SLAM heavily relies on manual parameter tuning for each sequence, which is considered impractical. Conversely, VIO-GO automatically fine-tunes the selected parameters across different environments. Furthermore, as explained in <xref ref-type="sec" rid="s4">Section 4</xref>, we opted for only eight key parameters that we identified as directly influencing the system performance. In contrast, Ultimate SLAM has a much larger set of parameters that can be adjusted for improved results, but this comes at the cost of requiring substantial computational time. For these reasons, we have grayed-out the Ultimate SLAM results in <xref ref-type="table" rid="T4">Table 4</xref>. This decision to downplay Ultimate SLAM was made to highlight the practical advantages of our simpler parameter set over the more computationally intensive Ultimate SLAM, thereby focusing on the efficiency and practicality of VIO-GO in real-world scenarios. Notably, VIO-GO8 surpasses all other approaches in 7 out of 10 sequences. With an average MPE of 0.36%, VIO-GO8 exhibits a 16% reduction in MPE compared to the 0.43% reported in (<xref ref-type="bibr" rid="B27">Rebecq et al., 2017b</xref>), a 33% lower MPE than EKLT-VIO (<xref ref-type="bibr" rid="B19">Mahlknecht et al., 2022</xref>) with 0.54% MPE, and an 86% lower MPE compared to EVIO (<xref ref-type="bibr" rid="B35">Zhu et al., 2017</xref>), which reports an MPE of 2.57%.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Performance of the proposed VIO-GO against other event-based VIO systems in terms of MPE in %. The USLAM&#x2a; results are obtained by individually tuning parameters for each sequence, whereas Fixed USLAM uses a single set of parameters that is tuned across all sequences simultaneously.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Dataset</th>
<th align="center">USLAM<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref> <xref ref-type="bibr" rid="B33">Vidal et al. (2018)</xref>
</th>
<th align="center">Fixed USLAM</th>
<th align="center">
<xref ref-type="bibr" rid="B27">Rebecq et al. (2017b)</xref>
</th>
<th align="center">EKLT-VIO <xref ref-type="bibr" rid="B19">Mahlknecht et al. (2022)</xref>
</th>
<th align="center">EVIO <xref ref-type="bibr" rid="B35">Zhu et al. (2017)</xref>
</th>
<th align="center">VIO-GO8 (8 parameters)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">boxes_6dof</td>
<td align="center">0.30</td>
<td align="center">0.68</td>
<td align="center">
<bold>0.36</bold>
</td>
<td align="center">0.84</td>
<td align="center">3.61</td>
<td align="center">0.41</td>
</tr>
<tr>
<td align="left">boxes_translation</td>
<td align="center">0.27</td>
<td align="center">1.12</td>
<td align="center">0.31</td>
<td align="center">0.48</td>
<td align="center">2.69</td>
<td align="center">
<bold>0.25</bold>
</td>
</tr>
<tr>
<td align="left">dynamic_6dof</td>
<td align="center">0.19</td>
<td align="center">0.76</td>
<td align="center">0.56</td>
<td align="center">0.79</td>
<td align="center">4.07</td>
<td align="center">
<bold>0.27</bold>
</td>
</tr>
<tr>
<td align="left">dynamic_translation</td>
<td align="center">0.18</td>
<td align="center">0.63</td>
<td align="center">0.39</td>
<td align="center">0.40</td>
<td align="center">1.90</td>
<td align="center">
<bold>0.25</bold>
</td>
</tr>
<tr>
<td align="left">hdr_boxes</td>
<td align="center">0.37</td>
<td align="center">1.01</td>
<td align="center">0.59</td>
<td align="center">0.46</td>
<td align="center">1.23</td>
<td align="center">
<bold>0.35</bold>
</td>
</tr>
<tr>
<td align="left">hdr_poster</td>
<td align="center">0.31</td>
<td align="center">1.48</td>
<td align="center">0.33</td>
<td align="center">0.65</td>
<td align="center">2.63</td>
<td align="center">
<bold>0.25</bold>
</td>
</tr>
<tr>
<td align="left">poster_6dof</td>
<td align="center">0.28</td>
<td align="center">0.59</td>
<td align="center">0.40</td>
<td align="center">
<bold>0.35</bold>
</td>
<td align="center">3.56</td>
<td align="center">0.50</td>
</tr>
<tr>
<td align="left">poster_translation</td>
<td align="center">0.12</td>
<td align="center">0.24</td>
<td align="center">0.46</td>
<td align="center">0.35</td>
<td align="center">0.94</td>
<td align="center">
<bold>0.23</bold>
</td>
</tr>
<tr>
<td align="left">shapes_6dof</td>
<td align="center">0.10</td>
<td align="center">1.07</td>
<td align="center">
<bold>0.42</bold>
</td>
<td align="center">0.60</td>
<td align="center">2.69</td>
<td align="center">0.77</td>
</tr>
<tr>
<td align="left">shapes_translation</td>
<td align="center">0.26</td>
<td align="center">1.36</td>
<td align="center">0.50</td>
<td align="center">0.51</td>
<td align="center">2.42</td>
<td align="center">
<bold>0.36</bold>
</td>
</tr>
<tr>
<td align="left">Average</td>
<td align="center">0.24</td>
<td align="center">0.89</td>
<td align="center">0.43</td>
<td align="center">0.54</td>
<td align="center">2.57</td>
<td align="center">
<bold>0.36</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>Requires substantial parameter adjustments based on the dynamic events in the scene.</p>
</fn>
<fn>
<p>The values displayed in bold show the best results.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>
<xref ref-type="fig" rid="F6">Figures 6</xref>&#x2013;<xref ref-type="fig" rid="F8">8</xref> illustrate heatmaps of the estimated trajectories from the proposed approach alongside the raw trajectory from Ultimate SLAM, both aligned with the ground truth trajectory for three different sequences from the Event Camera Dataset. In <xref ref-type="fig" rid="F6">Figure 6</xref>, which corresponds to the boxes_6dof sequence, VIO-GO8 demonstrates high trajectory accuracy, closely aligning with the ground truth and achieving a low APE. However, its APE is slightly higher than that of Ultimate SLAM. This difference is primarily attributed to Ultimate SLAM relying on extensive manual tuning across a wide range of parameters. In contrast, VIO-GO automatically optimizes a fixed subset of eight key parameters. While broader manual tuning can improve accuracy, it increases system complexity and limits scalability. VIO-GO prioritizes efficiency and generalizability by eliminating the need for manual intervention. <xref ref-type="fig" rid="F7">Figures 7</xref>, <xref ref-type="fig" rid="F8">8</xref> present results for the <italic>hdr_boxes</italic> and <italic>boxes_translation</italic> sequences, respectively. In both cases, VIO-GO8 outperforms Ultimate SLAM by producing trajectories that more closely align with the ground truth and achieving lower APE values. These improvements highlight VIO-GO&#x2019;s ability to adapt parameter configurations to challenging conditions without requiring manual tuning. The results further demonstrate the robustness and flexibility of the proposed framework across diverse scenarios.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Heatmaps presents the APE for the VIO-GO trajectory and the Ultimate SLAM raw trajectory for the <inline-formula id="inf119">
<mml:math id="m130">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mn>6</mml:mn>
<mml:mi>d</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> sequence, both aligned with the ground truth trajectory using a 6-DOF transformation in SE3 over a 5-s duration, as generated by the EVO tool. <bold>(a)</bold> VIO-GO evaluated on <italic>Boxes_6dof</italic>. <bold>(b)</bold> USLAM&#x2a; evaluated on <italic>Boxes_6dof</italic>.</p>
</caption>
<graphic xlink:href="frobt-12-1541017-g006.tif">
<alt-text content-type="machine-generated">Side-by-side plots showing Absolute Pose Error (APE) for translation with SE(3) Umeyama alignment. Plot (a) evaluates VIO-GO and (b) evaluates USLAM&#x2a; on Boxes_6dof. Both graphs use the same scale, with reference paths in dashed lines and color-coded error values from blue to red indicating the magnitude of deviation from the reference path. The axes represent x (meters) and z (meters).</alt-text>
</graphic>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Heatmaps presents the APE for the VIO-GO trajectory and the Ultimate SLAM raw trajectory for the <inline-formula id="inf120">
<mml:math id="m131">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> sequence, both aligned with the ground truth trajectory using a 6-DOF transformation in SE3 over a 5-s duration, as generated by the EVO tool. <bold>(a)</bold> VIO-GO evaluated on <italic>Boxes_translation</italic>. <bold>(b)</bold> USLAM&#x2a; evaluated on <italic>Boxes_translation</italic>.</p>
</caption>
<graphic xlink:href="frobt-12-1541017-g007.tif">
<alt-text content-type="machine-generated">Two graphs in a side-by-side comparison show APE with respect to translation in meters using SE(3) Umeyama alignment. Both plots feature curving lines in various colors. Graph (a) VIO-GO and graph (b) USLAM&#x2a; are evaluated on Boxes_translation. The background grid displays x-axis from -0.2 to 0.4 meters and z-axis from 1.0 to 1.7 meters, with a color bar ranging from approximately -0.066 to 0.066 on the right side to indicate error levels.</alt-text>
</graphic>
</fig>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Heatmaps presents the APE for the VIO-GO trajectory and the Ultimate SLAM raw trajectory for the <inline-formula id="inf121">
<mml:math id="m132">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>r</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>B</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> sequence, both aligned with the ground truth trajectory using a 6-DOF transformation in SE3 over a 5-s duration, as generated by the EVO tool. <bold>(a)</bold> VIO-GO evaluated on <italic>Hdr_boxes</italic>. <bold>(b)</bold> USLAM&#x2a; evaluated on <italic>Hdr_boxes</italic>.</p>
</caption>
<graphic xlink:href="frobt-12-1541017-g008.tif">
<alt-text content-type="machine-generated">Two graphs compare Absolute Pose Error (APE) with respect to translation in meters using SE(3) Umeyama alignment. Both graphs plot &#x22;z&#x22; against &#x22;x&#x22; with a reference trajectory in dashed gray. The left graph shows &#x22;VIO-GO evaluated on Hdr_boxes,&#x22; and the right shows &#x22;USLAM&#x2a; evaluated on Hdr_boxes.&#x22; Color gradients indicate error magnitudes, with values increasing from blue to red.</alt-text>
</graphic>
</fig>
<p>In <xref ref-type="fig" rid="F9">Figure 9</xref>, we employ relative error metrics to compare VIO-GO8 to Ultimate SLAM with its default parameter configuration applied to the <italic>hdr_boxes</italic> and <italic>boxes_translation</italic> sequences. The results show that VIO-GO8 notably reduces drift in the estimated trajectory over time. <xref ref-type="table" rid="T5">Table 5</xref> presents a time analysis comparison between Ultimate SLAM, using its default parameters, and VIO-GO8 with its optimal parameter set. As shown, VIO-GO8 requires significantly less time to process all datasets compared to Ultimate SLAM. This performance improvement is attributed to VIO-GO8&#x2019;s ability to dynamically select the best parameter set for each sequence, thereby reducing processing overhead in both the front-end and back-end stages. Furthermore, as previously discussed, VIO-GO8 outperforms the fixed parameter set approach by achieving an average MPE that is 58% lower than Ultimate SLAM&#x2019;s default parameters across all sequences.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>The relative error comparison between Ultimate SLAM with its default parameters and VIO-GO8 with its optimized parameters. <bold>(a)</bold> <italic>Hdr_boxes</italic> sequence. <bold>(b)</bold> <italic>Boxes_translation</italic> sequence.</p>
</caption>
<graphic xlink:href="frobt-12-1541017-g009.tif">
<alt-text content-type="machine-generated">Two box plots compare translational errors over time for two sequences with different parameters. The top plot shows the Hdr_boxes sequence, and the bottom shows the Boxes_translation sequence. Each plot presents translational error in meters on the y-axis against time in seconds on the x-axis. Red bars represent default parameters, and blue bars represent VIO-GO8 parameters. Error generally decreases over time in both plots, with VIO-GO8 parameters showing consistently lower error values than the default parameters.</alt-text>
</graphic>
</fig>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Elapsed time comparison between Ultimate SLAM with its default parameters and VIO-GO8 with its optimized parameters.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Dataset</th>
<th colspan="2" align="center">USLAM (default parameters)</th>
<th colspan="2" align="center">VIO-GO8</th>
</tr>
<tr>
<th align="center">Time cost (s)</th>
<th align="center">MPE (%)</th>
<th align="center">Time cost (s)</th>
<th align="center">MPE (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">boxes_6dof</td>
<td align="center">20.01</td>
<td align="center">0.49</td>
<td align="center">
<bold>18.56</bold>
</td>
<td align="center">
<bold>0.41</bold>
</td>
</tr>
<tr>
<td align="left">boxes_translation</td>
<td align="center">22.11</td>
<td align="center">0.38</td>
<td align="center">
<bold>20.30</bold>
</td>
<td align="center">
<bold>0.25</bold>
</td>
</tr>
<tr>
<td align="left">dynamic_6dof</td>
<td align="center">17.83</td>
<td align="center">0.66</td>
<td align="center">
<bold>14.50</bold>
</td>
<td align="center">
<bold>0.27</bold>
</td>
</tr>
<tr>
<td align="left">dynamic_translation</td>
<td align="center">17.65</td>
<td align="center">1.07</td>
<td align="center">
<bold>10.94</bold>
</td>
<td align="center">
<bold>0.25</bold>
</td>
</tr>
<tr>
<td align="left">hdr_boxes</td>
<td align="center">20.58</td>
<td align="center">1.12</td>
<td align="center">
<bold>18.39</bold>
</td>
<td align="center">
<bold>0.36</bold>
</td>
</tr>
<tr>
<td align="left">hdr_poster</td>
<td align="center">21.47</td>
<td align="center">0.51</td>
<td align="center">
<bold>18.83</bold>
</td>
<td align="center">
<bold>0.25</bold>
</td>
</tr>
<tr>
<td align="left">poster_6dof</td>
<td align="center">22.14</td>
<td align="center">0.96</td>
<td align="center">
<bold>21.90</bold>
</td>
<td align="center">
<bold>0.50</bold>
</td>
</tr>
<tr>
<td align="left">poster_translation</td>
<td align="center">18.71</td>
<td align="center">0.35</td>
<td align="center">
<bold>16.18</bold>
</td>
<td align="center">
<bold>0.23</bold>
</td>
</tr>
<tr>
<td align="left">shapes_6dof</td>
<td align="center">16.45</td>
<td align="center">1.46</td>
<td align="center">
<bold>11.30</bold>
</td>
<td align="center">
<bold>0.77</bold>
</td>
</tr>
<tr>
<td align="left">shapes_translation</td>
<td align="center">15.35</td>
<td align="center">0.70</td>
<td align="center">
<bold>12.99</bold>
</td>
<td align="center">
<bold>0.36</bold>
</td>
</tr>
<tr>
<td align="left">Average</td>
<td align="center">19.23</td>
<td align="center">0.77</td>
<td align="center">
<bold>16.39</bold>
</td>
<td align="center">
<bold>0.36</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The values displayed in bold show the best results.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s6">
<label>6</label>
<title>Discussion</title>
<p>In this section, we highlight the effectiveness of VIO-GO in addressing key challenges such as parameter optimization and computational efficiency. Additionally, we reflect on the broader impact of our findings, explore potential research avenues, and identify areas for improvement to guide future advancements in this field.</p>
<sec id="s6-1">
<label>6.1</label>
<title>Contributions</title>
<p>The primary contribution of VIO-GO is its ability to automatically optimize parameters for event-based VIO systems, significantly improving both accuracy and computational efficiency. Specifically, the VIO-GO8 approach, which optimizes eight key parameters, achieves an average MPE of 0.36%, outperforming fixed-parameter approaches such as Ultimate SLAM and other state-of-the-art methods, including EKLT-VIO and EVIO. These results underline the effectiveness of adaptive parameter tuning in enhancing VIO performance across diverse and dynamic environments.</p>
<p>A critical observation is the scalability of VIO-GO, where system performance improves with the inclusion of additional optimized parameters. For instance, a comparison between VIO-GO2 and VIO-GO8 demonstrates the benefits of comprehensive parameter optimization. Moreover, VIO-GO eliminates the need for manual parameter tuning required by previous methods, significantly reducing deployment time and effort. This makes it particularly well-suited for applications in Industry 4.0, where environments are highly variable and demand rapid adaptation. The core design of VIO-GO emphasizes generalizability. By dynamically optimizing a fixed set of key parameters based on scene characteristics, it adapts automatically to diverse conditions without relying on predefined configurations. In contrast to traditional motion-compensation approaches that require extensive manual adjustment for each new environment, VIO-GO offers a more scalable and practical solution.</p>
<p>In comparison with existing approaches, VIO-GO introduces a paradigm shift by automating the parameter optimization process. Our results show that VIO-GO significantly reduces trajectory drift over time and achieves a lower APE compared to fixed-parameter approaches. This is crucial for real-time applications, making VIO-GO an ideal candidate for resource-constrained scenarios in industrial robotics and autonomous navigation.</p>
</sec>
<sec id="s6-2">
<label>6.2</label>
<title>Limitations</title>
<p>While VIO-GO demonstrates promising results, several limitations remain. One key challenge is its dependency on a predefined set of key parameters, which may constrain its adaptability to highly diverse or previously unseen environments. Future iterations could expand the parameter set or incorporate environment-specific variables, allowing the system to adapt more effectively to complex scenarios. Another limitation is the sensitivity of the system to initial conditions, such as IMU bias and feature selection, which may affect stability during extended operations. Future efforts could address these challenges through advanced initialization methods and noise mitigation strategies.</p>
<p>Additionally, while the Event Camera Dataset provides a valuable and well-calibrated benchmark for evaluating event-based VIO systems, it represents a relatively controlled environment. In real-world scenarios, factors like unstructured environments, sensor noise, and erratic motion patterns can significantly affect event data quality. VIO-GO is designed to address such variability through its core capability of dynamically optimizing key system parameters based on the characteristics of each scene. This allows the system to adapt in real time without requiring manual reconfiguration. Nevertheless, transferring the system from a controlled dataset to real-world deployment may affect the effectiveness of the selected parameter sets. Real-world conditions could present edge cases or variations not fully represented in the dataset, potentially impacting the convergence behavior or responsiveness of the optimization process. For instance, parameters such as the frame size and noise event rate might need adjustments to account for fluctuating event densities caused by background activity. Furthermore, parameters related to feature extraction may need to be tuned to handle less structured or more repetitive textures commonly found in natural scenes. These factors underscore that testing VIO-GO in real-world environments would provide a deeper understanding of its robustness in diverse and unpredictable conditions. Lastly, the use of BGD for parameter optimization, while effective, could be complemented by exploring alternative techniques, such as SGD, Bayesian optimization, or Gauss-Newton methods, to improve convergence speed and efficiency.</p>
</sec>
<sec id="s6-3">
<label>6.3</label>
<title>Future directions</title>
<p>Building on the current success of VIO-GO, several promising research avenues could extend its capabilities:<list list-type="order">
<list-item>
<p>Expansion of the Parameter Optimization Scope: Extending the optimization to a larger set of parameters remains a promising avenue for future work. While this study limited the number of optimized parameters to maintain practical feasibility, expanding this scope could potentially unlock additional performance improvements.</p>
</list-item>
<list-item>
<p>Integration with Other Event-Based SLAM Approaches: Future work could explore extending VIO-GO to integrate with other event-based SLAM systems. This would help develop more robust solutions adaptable to a wider range of applications.</p>
</list-item>
<list-item>
<p>Exploration of Advanced Event Processing Techniques: Future studies could look into advanced event-based processing techniques, including deep learning-based methods for event-to-image conversion or more sophisticated feature tracking approaches. These could further boost the performance of event-based VIO systems.</p>
</list-item>
<list-item>
<p>Real-Time Adaptation and On-the-Fly Tuning: Implementing real-time adaptation and on-the-fly parameter tuning would make VIO-GO more suitable for autonomous systems operating in unpredictable environments, minimizing the need for pre-set parameters.</p>
</list-item>
</list>
</p>
<p>By addressing these limitations and expanding the scope of the study, future research could significantly advance the field of Event SLAM, contributing to the development of more robust, efficient, and adaptable systems for autonomous navigation in dynamic environments.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s7">
<label>7</label>
<title>Conclusion</title>
<p>This work presents VIO-GO, a novel framework for automated parameter optimization in event-based VIO systems, tailored for use in dynamic environments central to Industry 4.0 applications. Designed to address the challenges of dynamic and variable environments, VIO-GO achieves a balance of accuracy and computational efficiency by using motion-compensated images and a BGD algorithm, enhancing the performance and robustness of Event SLAM systems.</p>
<p>Our evaluation on the Event Camera Dataset shows that VIO-GO outperforms fixed-parameter approaches, achieving a 60% reduction in MPE. The system successfully identifies optimal parameters for Ultimate SLAM across multiple sequences, confirming its adaptability to scenarios characterized by fluctuating event rates. This capability is particularly critical for industrial applications, where environmental variability demands highly responsive and efficient navigation solutions.</p>
<p>These results highlight the importance of automated parameter optimization in event-based SLAM systems. Future research should focus on testing VIO-GO in more diverse and complex real-world settings, incorporating advanced event-based processing techniques and exploring alternative optimization methods to further enhance performance. Additionally, VIO-GO&#x2019;s adaptability can be further evaluated across a wider range of datasets and integrated with other event-based SLAM approaches beyond Ultimate SLAM, expanding its applicability and generalizability to real-world scenarios. By addressing these directions, VIO-GO could establish a new standard for robust, scalable, and adaptive SLAM solutions, particularly in the demanding contexts of Industry 4.0 and beyond.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s8">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>SS: Conceptualization, Investigation, Methodology, Software, Validation, Writing &#x2013; original draft, Writing &#x2013; review and editing. AS: Conceptualization, Investigation, Validation, Writing &#x2013; original draft, Writing &#x2013; review and editing. JM: Conceptualization, Investigation, Project administration, Supervision, Visualization, Writing &#x2013; review and editing. BA: Conceptualization, Investigation, Supervision, Writing &#x2013; review and editing. NM: Conceptualization, Investigation, Supervision, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="COI-statement" id="s11">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s12">
<title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. Generative AI was exclusively used to correct grammar and language mistakes.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s13">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2686736/overview">Rajkumar Muthusamy</ext-link>, Dubai Future Foundation, United Arab Emirates</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2170680/overview">Weibin Guo</ext-link>, Chinese Academy of Sciences (CAS), China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2947013/overview">Chuanfei Hu</ext-link>, Southeast University, China</p>
</fn>
</fn-group>
<fn-group>
<fn id="n1">
<label>1</label>
<p>Refer to <xref ref-type="table" rid="T1">Table 1</xref> for key parameters used in the state estimation framework and optimized through the proposed method outlined in <xref ref-type="sec" rid="s4">Section 4</xref>.</p>
</fn>
<fn id="n2">
<label>2</label>
<p>IMU datasheet: <ext-link ext-link-type="uri" xlink:href="https://www.cdiweb.com/datasheets/invensense/ps-mpu-6100a">https://www.cdiweb.com/datasheets/invensense/ps-mpu-6100a</ext-link>
</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Alzugaray</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Chli</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Ace: an efficient asynchronous corner tracker for event cameras</article-title>,&#x201d; in <source>2018 international conference on 3D vision</source>, <fpage>653</fpage>&#x2013;<lpage>661</lpage>. <pub-id pub-id-type="doi">10.1109/3DV.2018.00080</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Beomsoo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ravankar</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Emaru</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Mobile robot navigation based on deep reinforcement learning with 2d-lidar sensor using stochastic approach</article-title>,&#x201d; in <source>2021 IEEE international conference on intelligence and safety for robotics (ISR)</source>, <fpage>417</fpage>&#x2013;<lpage>422</lpage>. <pub-id pub-id-type="doi">10.1109/ISR50024.2021.9419565</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Campos</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Elvira</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rodr&#xed;guez</surname>
<given-names>J. J. G.</given-names>
</name>
<name>
<surname>Montiel</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Tard&#xf3;s</surname>
<given-names>J. D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Orb-slam3: an accurate open-source library for visual, visual&#x2013;inertial, and multimap slam</article-title>. <source>IEEE Trans. Robotics</source> <volume>37</volume>, <fpage>1874</fpage>&#x2013;<lpage>1890</lpage>. <pub-id pub-id-type="doi">10.1109/tro.2021.3075644</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Censi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Low-latency event-based visual odometry</article-title>,&#x201d; in <source>2014 IEEE international conference on robotics and automation (ICRA)</source>. <pub-id pub-id-type="doi">10.1109/icra.2014.6906931</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chandra</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ragan-Kelley</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Meijer</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Gradient descent: the ultimate optimizer</article-title>
</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Esvio: event-based stereo visual inertial odometry</article-title>. <source>IEEE Robot. Autom. Lett.</source> <volume>8</volume>, <fpage>3661</fpage>&#x2013;<lpage>3668</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2023.3269950</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gallego</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Delbruck</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Orchard</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bartolozzi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Taba</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Censi</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Event-based vision: a survey</article-title>. <source>IEEE Trans. Pattern Analysis &#x26; Mach. Intell.</source> <volume>44</volume>, <fpage>154</fpage>&#x2013;<lpage>180</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2020.3008413</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gehrig</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Rebecq</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gallego</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>EKLT: asynchronous photometric feature tracking using events and frames</article-title>. <source>Int. J. Comput. Vis.</source> <volume>128</volume>, <fpage>601</fpage>&#x2013;<lpage>618</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-019-01209-w</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grupp</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Evo: python package for the evaluation of odometry and slam</article-title>.</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Pl-evio: robust monocular event-based visual inertial odometry with point and line features</article-title>. <source>IEEE Trans. Automation Sci. Eng.</source> <volume>21</volume>, <fpage>6277</fpage>&#x2013;<lpage>6293</lpage>. <pub-id pub-id-type="doi">10.1109/TASE.2023.3324365</pub-id>
</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Hadviger</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>&#x160;tironja</surname>
<given-names>V.-J.</given-names>
</name>
<name>
<surname>Cvi&#x161;i&#x107;</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Markovi&#x107;</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Vra&#x17e;i&#x107;</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Petrovi&#x107;</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Stereo visual localization dataset featuring event cameras</article-title>,&#x201d; in <source>2023 European conference on Mobile robots (ECMR)</source>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1109/ECMR59166.2023.10256407</pub-id>
</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tao</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Event-based simultaneous localization and mapping: a comprehensive survey</article-title>
</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Leutenegger</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Davison</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Real-time 3d reconstruction and 6-dof tracking with an event camera</article-title>. <source>Comput. Vis. &#x2013; ECCV</source> <volume>2016</volume>, <fpage>349</fpage>&#x2013;<lpage>364doi</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-46466-4_21</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Kueng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mueggler</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Gallego</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Low-latency visual odometry using event-based feature tracks</article-title>,&#x201d; in <source>2016 IEEE/RSJ international conference on intelligent robots and systems (IROS)</source>. <pub-id pub-id-type="doi">10.1109/iros.2016.7758089</pub-id>
</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leutenegger</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Furgale</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rabaud</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Chli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Konolige</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Siegwart</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Keyframe-based visual-inertial slam using nonlinear optimization</article-title>. <source>Robotics Sci. Syst. IX</source>. <pub-id pub-id-type="doi">10.15607/rss.2013.ix.037</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ang</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Rus</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Online localization with imprecise floor space maps using stochastic gradient descent</article-title>,&#x201d; in <source>2020 IEEE/RSJ international conference on intelligent Robots and Systems (IROS)</source>, <fpage>8571</fpage>&#x2013;<lpage>8578</lpage>. <pub-id pub-id-type="doi">10.1109/IROS45743.2020.9340793</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Lucas</surname>
<given-names>B. D.</given-names>
</name>
<name>
<surname>Kanade</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1981</year>). &#x201c;<article-title>An iterative image registration technique with an application to stereo vision</article-title>,&#x201d; in <source>Proceedings of the 7th international joint conference on artificial intelligence - volume 2</source> (<publisher-loc>San Francisco, CA, USA</publisher-loc>: <publisher-name>Morgan Kaufmann Publishers Inc.), IJCAI&#x2019;81</publisher-name>), <fpage>674</fpage>&#x2013;<lpage>679</lpage>.</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>K.-T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Real-time dense monocular slam with online adapted depth prediction network</article-title>. <source>IEEE Trans. Multimedia</source> <volume>21</volume>, <fpage>470</fpage>&#x2013;<lpage>483</lpage>. <pub-id pub-id-type="doi">10.1109/TMM.2018.2859034</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahlknecht</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gehrig</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nash</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rockenbauer</surname>
<given-names>F. M.</given-names>
</name>
<name>
<surname>Morrell</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Delaune</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Exploring event camera-based odometry for planetary robots</article-title>. <source>IEEE Robot. Autom. Lett.</source> <volume>7</volume>, <fpage>8651</fpage>&#x2013;<lpage>8658</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2022.3187826</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mueggler</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Huber</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Event-based, 6-dof pose tracking for high-speed maneuvers</article-title>,&#x201d; in <source>2014 IEEE/RSJ international conference on intelligent robots and systems</source>. <pub-id pub-id-type="doi">10.1109/iros.2014.6942940</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mueggler</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Gallego</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rebecq</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017a</year>). <article-title>Continuous-time visual-inertial trajectory estimation with event cameras</article-title>. <source>
<italic>Corr.</italic> abs/1702</source>. <pub-id pub-id-type="doi">10.1109/TRO.2018.2858287</pub-id>
</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mueggler</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Rebecq</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gallego</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Delbruck</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017b</year>). <article-title>The event-camera dataset and simulator: event-based data for pose estimation, visual odometry, and slam</article-title>. <source>Int. J. Robotics Res.</source> <volume>36</volume>, <fpage>142</fpage>&#x2013;<lpage>149</lpage>. <pub-id pub-id-type="doi">10.1177/0278364917691115</pub-id>
</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mueggler</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Gallego</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rebecq</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Continuous-time visual-inertial odometry for event cameras</article-title>. <source>IEEE Trans. Robotics</source> <volume>34</volume>, <fpage>1425</fpage>&#x2013;<lpage>1440</lpage>. <pub-id pub-id-type="doi">10.1109/tro.2018.2858287</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mustapha</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mohamed</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>An overview of gradient descent algorithm optimization in machine learning: application in the ophthalmology field</article-title>,&#x201d; in <source>Smart applications and data analysis</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Hamlich</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bellatreche</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Mondal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ordonez</surname>
<given-names>C.</given-names>
</name>
</person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>349</fpage>&#x2013;<lpage>359</lpage>.</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Nartey</surname>
<given-names>O. T.</given-names>
</name>
<name>
<surname>Jan</surname>
<given-names>S. U.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Relevance gradient descent for parameter optimization of image enhancement</article-title>. <source>Comput. &#x26; Graph.</source> <volume>117</volume>, <fpage>124</fpage>&#x2013;<lpage>133</lpage>. <pub-id pub-id-type="doi">10.1016/j.cag.2023.10.016</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rebecq</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Horstschaefer</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Gallego</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017a</year>). <article-title>Evo: a geometric approach to event-based 6-dof parallel tracking and mapping in real time</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>2</volume>, <fpage>593</fpage>&#x2013;<lpage>600</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2016.2645143</pub-id>
</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rebecq</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Horstschaefer</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017b</year>). &#x201c;<article-title>Real-time visual-inertial odometry for event cameras using keyframe-based nonlinear optimization</article-title>. <pub-id pub-id-type="doi">10.5244/c.31.16</pub-id>
</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rosten</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Drummond</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Machine learning for high-speed corner detection</article-title>. <source>Comput. Vis. &#x2013; ECCV</source> <volume>2006</volume>, <fpage>430</fpage>&#x2013;<lpage>443doi</lpage>. <pub-id pub-id-type="doi">10.1007/11744023_34</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sahili</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Hassan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sakhrieh</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Mounsef</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Maalouf</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Arain</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A survey of visual slam methods</article-title>. <source>IEEE Access</source> <volume>11</volume>, <fpage>139643</fpage>&#x2013;<lpage>139677</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2023.3341489</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>N. S.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Performance analysis of large scale machine learning optimization algorithms</article-title>,&#x201d; in <source>2023 IEEE 12th international conference on communication systems and network technologies (CSNT)</source>, <fpage>226</fpage>&#x2013;<lpage>230</lpage>. <pub-id pub-id-type="doi">10.1109/CSNT57126.2023.10134605</pub-id>
</mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Research on slam algorithm of mobile robot vision based on deep learning</article-title>,&#x201d; in <source>2021 global reliability and prognostics and health management (PHM-Nanjing)</source>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1109/PHM-Nanjing52125.2021.9612660</pub-id>
</mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Torroba</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Cella</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ter&#xe1;n</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rolleberg</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Folkesson</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Online stochastic variational gaussian process mapping for large-scale bathymetric slam in real time</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>8</volume>, <fpage>3150</fpage>&#x2013;<lpage>3157</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2023.3264750</pub-id>
</mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vidal</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Rebecq</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Horstschaefer</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Ultimate slam? Combining events, images, and imu for robust visual slam in hdr and high-speed scenarios</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>3</volume>, <fpage>994</fpage>&#x2013;<lpage>1001</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2018.2793357</pub-id>
</mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Research on event accumulator settings for event-based slam</article-title>,&#x201d; in <source>2022 6th international conference on robotics, control and automation (ICRCA)</source>, <fpage>50</fpage>&#x2013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1109/ICRCA55033.2022.9828933</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>A. Z.</given-names>
</name>
<name>
<surname>Atanasov</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Daniilidis</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Event-based visual inertial odometry</article-title>,&#x201d; in <source>2017 IEEE conference on computer vision and pattern recognition (CVPR)</source>. <pub-id pub-id-type="doi">10.1109/cvpr.2017.616</pub-id>
</mixed-citation>
</ref>
</ref-list>
</back>
</article>