<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Comput. Sci.</journal-id>
<journal-title>Frontiers in Computer Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Comput. Sci.</abbrev-journal-title>
<issn pub-type="epub">2624-9898</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcomp.2025.1510252</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Computer Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>MoNetViT: an efficient fusion of CNN and transformer technologies for visual navigation assistance with multi query attention</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Triyono</surname> <given-names>Liliek</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2864703/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Gernowo</surname> <given-names>Rahmat</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Prayitno</surname></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Doctoral Program of Information System, Diponegoro University</institution>, <addr-line>Central Java, Semarang</addr-line>, <country>Indonesia</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Electrical Engineering, Politeknik Negeri Semarang</institution>, <addr-line>Semarang</addr-line>, <country>Indonesia</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: Sokratis Makrogiannis, Delaware State University, United States</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: Karisma Putra, Muhammadiyah University of Yogyakarta, Indonesia</p>
<p>Mosiur Rahaman, Asia University, Taiwan</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Liliek Triyono, <email>liliek.triyono@polines.ac.id</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>10</day>
<month>02</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>7</volume>
<elocation-id>1510252</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>01</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Triyono, Gernowo and Prayitno.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Triyono, Gernowo and Prayitno</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Aruco markers are crucial for navigation in complex indoor environments, especially for those with visual impairments. Traditional CNNs handle image segmentation well, but transformers excel at capturing long-range dependencies, essential for machine vision tasks. Our study introduces MoNetViT (Mini-MobileNet MobileViT), a lightweight model combining CNNs and MobileViT in a dual-path encoder to optimize global and spatial image details. This design reduces complexity and boosts segmentation performance. The addition of a multi-query attention (MQA) module enhances multi-scale feature integration, allowing end-to-end learning guided by ground truth. Experiments show MoNetViT outperforms other semantic segmentation algorithms in efficiency and effectiveness, particularly in detecting Aruco markers, making it a promising tool to improve navigation aids for the visually impaired.</p>
</abstract>
<kwd-group>
<kwd>indoor navigation</kwd>
<kwd>computer vision</kwd>
<kwd>markers</kwd>
<kwd>assistive technology</kwd>
<kwd>mobile devices</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="4"/>
<equation-count count="30"/>
<ref-count count="53"/>
<page-count count="14"/>
<word-count count="10437"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computer Vision</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>Navigating independently is a major challenge for individuals with visual impairments. It affects their ability to perform daily tasks and limits their involvement in social and economic activities. This challenge reduces their personal autonomy and impacts their overall quality of life. Indoor navigation issues for visually impaired individuals persist, as current solutions frequently do not overcome critical limits. Conventional GPS is inadequate inside due to signal interference, requiring alternative systems such as beacon-based technologies and smartphone applications that employ digital maps (<xref ref-type="bibr" rid="ref40">Theodorou et al., 2022</xref>; <xref ref-type="bibr" rid="ref20">Kubota, 2024</xref>). Nonetheless, these systems frequently depend on pre-existing maps, which are not universally accessible, hence constraining their efficacy (<xref ref-type="bibr" rid="ref20">Kubota, 2024</xref>).</p>
<p>The intricacy of interior environments exacerbates the problem, as visually impaired individuals encounter challenges in traversing unfamiliar settings due to ambiguous aural or tactile signals and various impediments (<xref ref-type="bibr" rid="ref17">Jeamwatthanachai et al., 2019</xref>; <xref ref-type="bibr" rid="ref14">Fernando et al., 2023</xref>). Although systems such as Snap&#x0026;Nav provide navigation solutions through the creation of node maps, their dependence on sighted aid diminishes user autonomy (<xref ref-type="bibr" rid="ref20">Kubota, 2024</xref>). Moreover, the computing requirements of contemporary models provide difficulties. Advanced algorithms and machine learning techniques enhance obstacle identification and route planning but frequently necessitate substantial processing resources, rendering them impractical for mobile applications (<xref ref-type="bibr" rid="ref38">Tao and Ganz, 2020</xref>; <xref ref-type="bibr" rid="ref36">Shah et al., 2023</xref>). The integration of IoT and cloud computing introduces additional complexity, emphasizing the necessity for lightweight, dependable systems designed for visually impaired users (<xref ref-type="bibr" rid="ref29">Messaoudi et al., 2020</xref>). Rectifying these deficiencies is essential for the advancement of inclusive indoor navigation solutions.</p>
<p>Recent research has increasingly focused on developing advanced navigation aids to support visually impaired individuals in both indoor and outdoor environments. Technologies such as deep learning, machine vision, wearable devices, and mobile applications have been leveraged to enhance navigation capabilities, offering promising solutions to this pervasive issue (<xref ref-type="bibr" rid="ref3">Bai et al., 2019</xref>; <xref ref-type="bibr" rid="ref11">El-taher et al., 2021</xref>; <xref ref-type="bibr" rid="ref21">Kuriakose et al., 2021</xref>; <xref ref-type="bibr" rid="ref26">Mart&#x00ED;nez-Cruz et al., 2021</xref>).</p>
<p>The importance of developing an effective navigation assistance system for visually impaired individuals, particularly in obstacle-filled indoor environments, cannot be overstated. These environments present unique challenges that require sophisticated solutions capable of providing accurate and real-time guidance. The integration of advanced AI models, such as MobileNetV2 and MobileViTV2, along with multi-query attention mechanisms, has shown potential in creating robust and efficient navigation systems. These models aim to empower visually impaired individuals, allowing them to navigate unfamiliar spaces with confidence and independence (<xref ref-type="bibr" rid="ref44">Wang et al., 2019</xref>).</p>
<p>The main research problem addressed in this study is the development of an independent navigation model that improves the accuracy and efficiency of detecting and interpreting navigation markers under extreme conditions for people with visual impairment. Traditional navigation systems often fall short in complex, obstacle-filled indoor environments, necessitating the need for a more advanced solution. This research proposes integrating MobileNetV2 and MobileViTV2 methods, enhanced by multi-query attention mechanisms, to develop a navigation model that provides precise and reliable assistance, thereby improving the quality of life for visually impaired individuals.</p>
<p>The integration of MobileNetV2 and MobileViTV2 methods represents a cutting-edge approach to developing an independent navigation model for the visually impaired. MobileNetV2, introduced by <xref ref-type="bibr" rid="ref35">Sandler et al. (2018)</xref>, is designed to operate efficiently on mobile and embedded devices, making it highly suitable for real-time applications. Its architecture employs inverted residuals and linear bottlenecks, which help maintain high accuracy while reducing computational demands. This efficiency is crucial for applications requiring portability and immediate response, such as navigation aids for visually impaired individuals.</p>
<p>On the other hand, MobileViTV2, as explored by <xref ref-type="bibr" rid="ref5">Chen et al. (2021)</xref>, utilizes vision transformers to enhance the model&#x2019;s capability to understand visual contexts. Vision transformers are adept at capturing long-range dependencies within images, providing a more comprehensive interpretation of complex scenes. The integration of these technologies, combined with multi-query attention mechanisms as highlighted by <xref ref-type="bibr" rid="ref27">Mehta and Apple (2022)</xref>, allows the model to focus on multiple aspects of the visual input simultaneously. This multifaceted attention mechanism is instrumental in improving the accuracy and timeliness of navigation instructions, thus providing a significant advancement over existing models.</p>
<p>Existing research on navigation aids for visually impaired individuals has explored a variety of technological solutions, ranging from GPS-based applications to wearable devices and computer vision techniques. Studies like those by <xref ref-type="bibr" rid="ref26">Mart&#x00ED;nez-Cruz et al. (2021)</xref> and <xref ref-type="bibr" rid="ref11">El-taher et al. (2021)</xref> have highlighted the limitations of GPS in indoor environments and the bulkiness of wearable devices, respectively. These limitations underscore the need for more refined and user-friendly solutions. The integration of deep learning models, such as those using CNNs, has shown promise; however, these models often require substantial computational resources, limiting their practicality in mobile settings (<xref ref-type="bibr" rid="ref3">Bai et al., 2019</xref>).</p>
<p>The recent development of efficient models like MobileNetV2 and vision transformers like MobileViTV2 addresses some of these challenges by offering high accuracy with reduced computational demands. However, a gap remains in the effective integration of these technologies to develop a comprehensive navigation system that is both lightweight and capable of real-time processing. Additionally, the potential benefits of multi-query attention mechanisms in enhancing the focus and accuracy of these models have not been fully explored. This gap presents an opportunity to develop a novel, integrated solution that leverages these advanced techniques for improved navigation assistance.</p>
<p>The aim of this research is to develop a new navigation model for visually impaired individuals by combining MobileNetV2 and MobileViTV2 with multi-query attention mechanisms. The hypothesis is that this AI model will improve both the accuracy and efficiency of Aruco marker detection under challenging conditions compared to existing models. This advancement is expected to offer reliable navigation assistance in indoor environments, enhancing the quality of life for visually impaired users. The study focuses on designing, implementing, and evaluating the model, with future work aimed at refining fusion mechanisms, reducing model complexity, and exploring transfer learning to maintain high accuracy while minimizing computational demands.</p>
<p>Following a brief introduction of the problem statement and the proposed method, the rest of the paper is structured as follows: Section 2 outlines the research methodology used to conduct the study. In Section 3, we present the search results obtained from the research. Section 4 discusses the findings related to multi-scale features and the various combinations of MQA and FFM used to improve model segmentation of multi-class ArUco markers. Finally, the conclusion of the paper is provided in Section 5.</p>
</sec>
<sec sec-type="methods" id="sec2">
<label>2</label>
<title>Methods</title>
<sec id="sec3">
<label>2.1</label>
<title>Transformer and CNNs</title>
<p>CNNs have demonstrated exceptional performance in various image segmentation tasks, showcasing their robust feature representation capabilities. However, despite these strengths, CNN-based methods frequently encounter limitations in modeling long-range relationships. A primary issue is their inefficiency in capturing global context information. Methods that rely on stacking receptive fields necessitate continuous downsampling convolution operations, leading to deeper networks. Training such deep neural networks on small datasets can present significant challenges, including training instability and overfitting. Overfitting is particularly common in deep learning models due to their strong expressive ability relative to traditional models (<xref ref-type="bibr" rid="ref48">Zhang et al., 2023</xref>). Non-local attention mechanisms have been increasingly utilized in various fields to address challenges related to capturing long-range dependencies and global information (<xref ref-type="bibr" rid="ref28">Mei et al., 2020</xref>; <xref ref-type="bibr" rid="ref16">Huang et al., 2022</xref>; <xref ref-type="bibr" rid="ref1">Abozeid et al., 2023</xref>; <xref ref-type="bibr" rid="ref53">Zhou et al., 2023</xref>). While these mechanisms can enhance the network&#x2019;s ability to capture global context, they also introduce considerable computational complexity. This complexity, which is quadratic in relation to the input size, often renders these methods impractical for high-resolution images.</p>
<p>Attention mechanisms were utilized in numerous research that focused on integrating Convolutional Neural Networks. Especially to further enhance the output processing of CNNs. Various visual tasks were implemented with integrated approaches, including video processing (<xref ref-type="bibr" rid="ref33">Qi and Zhang, 2023</xref>; <xref ref-type="bibr" rid="ref37">Sun et al., 2022</xref>; <xref ref-type="bibr" rid="ref31">Mujtaba et al., 2022</xref>), image classification (<xref ref-type="bibr" rid="ref10">Dosovitskiy et al., 2021</xref>; <xref ref-type="bibr" rid="ref25">Liu et al., 2021</xref>), and object detection (<xref ref-type="bibr" rid="ref4">Benmouna et al., 2023</xref>; <xref ref-type="bibr" rid="ref45">Wen et al., 2023</xref>).</p>
<p>The transformer in natural language processing used transformation tasks (<xref ref-type="bibr" rid="ref41">Vaswani et al., 2023</xref>). Several natural language processing activities have since shifted to using it. Some natural language processing activities have switched to using ViT. Pre-training on very large datasets is required for ViT (<xref ref-type="bibr" rid="ref6">Chen et al., 2023</xref>; <xref ref-type="bibr" rid="ref30">Misawa et al., 2024</xref>). To the State of the Art in the natural image segmentation task, Imagenet replaced the encoder component of the decoding network with a transformer (<xref ref-type="bibr" rid="ref8">Doppalapudi, 2023</xref>; <xref ref-type="bibr" rid="ref46">Xia and Kim, 2023</xref>).</p>
<p>Although transformer-based models have demonstrated impressive skills in diverse visual tasks, they have not yet attained acceptable results when compared to traditional CNNs. Transformer designs still exhibit worse performance in visual tasks compared to similarly-sized CNNs, such as EfficientNet (<xref ref-type="bibr" rid="ref39">Thakur et al., 2023</xref>). The computational cost of transformers based on the mechanism of self-attention is <inline-formula>
<mml:math id="M1">
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mi mathvariant="normal">C</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, in contrast to the convolution-based CNNs <inline-formula>
<mml:math id="M2">
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi mathvariant="normal">C</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="ref50">Zhou et al., 2024</xref>). Therefore, employing the transformer for image-related activities will unavoidably need a substantial amount of GPU resources.</p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Image segmentation using transformer and CNNs</title>
<p>The present cutting-edge architecture in computer vision predominantly depends on complete CNNs, with UNet (<xref ref-type="bibr" rid="ref5">Chen et al., 2021</xref>) and its variations being notable instances. The current state-of-the-art (SOTA) framework in computer vision primarily relies on full CNNs, with UNet and its variants being prominent examples. UNet (<xref ref-type="bibr" rid="ref5">Chen et al., 2021</xref>) employs an encoding-decoding network architecture. This architecture utilizes cascaded convolutional layers to extract various levels of visual characteristics. The decoder utilizes skip connections to recycle high-resolution feature maps generated by the encoder, enabling the retrieval of crucial feature information (<xref ref-type="bibr" rid="ref32">Petit et al., 2021</xref>).</p>
</sec>
<sec id="sec5">
<label>2.3</label>
<title>Lightweight networks</title>
<p>Deep learning, although powerful, often requires extensive training data to effectively enhance model learning. However, challenges arise in scenarios like the ArUco dataset due to limitations in data collection related to factors such as lighting conditions, capture angles, and distances. Moreover, the availability of large, publicly accessible datasets is limited, further complicating model training (<xref ref-type="bibr" rid="ref24">Lee et al., 2019</xref>). To address these challenges, the development of lightweight deep learning models becomes imperative.</p>
<p>Research in deep learning has demonstrated that supervised training of deep learning models heavily relies on large labeled datasets (<xref ref-type="bibr" rid="ref18">Karimi et al., 2020</xref>). This requirement poses a significant challenge, especially in scenarios where data collection is constrained. Techniques such as model optimization, pruning, quantization, and knowledge distillation have been explored to create lightweight deep-learning models suitable for mobile terminals (<xref ref-type="bibr" rid="ref42">Wang et al., 2022</xref>). These approaches aim to reduce the computational burden while maintaining model performance.</p>
<p>A self-attention-based vision transformer (ViT), known as MobileViT, is employed to learn the global representation of images. MobileViT (<xref ref-type="bibr" rid="ref27">Mehta and Apple, 2022</xref>) stands out as the initial lightweight, general-purpose transformer designed for mobile devices. An approach integrating a transformer with a CNN-based lightweight model was investigated, with a particular focus on assessing the feasibility of this lightweight network model for the challenging task of ArUco marker segmentation.</p>
</sec>
<sec id="sec6">
<label>2.4</label>
<title>Mini-MobileNet-MobileViT network</title>
<p>In this part, the Mini-MobileNet-MobileViT (MoNetViT) network architecture and its principal network components were introduced. The system&#x2019;s backbone structure follows to the architecture of an encoder and decoder, as represented in <xref ref-type="fig" rid="fig1">Figure 1A</xref>. Section 2.4.1 offers a more detailed explanation sub-network of the encoder, while Section 2.4.2 focuses on the exploration sub-network of the decoder. The MobileViT module, a crucial part of the network&#x2019;s encoder architecture, is introduced in Section 2.4.3. This section covers the architecture of the MobileViT module, its primary calculation process internally, and the comparisons between this module and CNN. Furthermore, the MQA module that is suggested in this study is described in Section 2.4.4. The Globalized Block and the Asymmetrical Globalized Block, as well as the justification for their adoption, are part of this module.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>MoNetViT Main Architectural diagram and the essential network components. <bold>(A)</bold> MoNetViT deep CNN encoder and a few basic modules <bold>(B)</bold> Feature Fusion Modul (FFM) <bold>(C)</bold> Illustration calculation between pixels in MobileViT <bold>(D)</bold> MobileViT-Block.</p>
</caption>
<graphic xlink:href="fcomp-07-1510252-g001.tif"/>
</fig>
<sec id="sec7">
<label>2.4.1</label>
<title>Encoder sub-network</title>
<p>The proposed model will be developed using an encoder-decoder structure, where the encoder will build two parallel paths connected by a series of attention additions, improving the model&#x2019;s ability to capture spatial and channel dependencies. The encoder will use MobileNet v2 (MN2 block) (<xref ref-type="bibr" rid="ref35">Sandler et al., 2018</xref>) and MobileViT block as the base module. <inline-formula>
<mml:math id="M3">
<mml:mi>I</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, is the representation of the input image, where <italic>H</italic> and &#x1d44a; stand for the input image&#x2019;s height and width, respectively. The input image undergoes resolution degradation through three consecutive stages. In each stage, the size of the feature map is reduced by a factor of 2. As a result, the output feature maps are reduced in size to one-half, one-fourth, and one-eighth of the initial feature map. The MobileViT block is one of the essential components used in the encoder. The input and output sizes of the MobileViT block are the same, indicating that this module does not change the spatial dimensions of the feature map. The MN2 block is another basic module used in the encoder. Stride 1 implies that the module does not perform resolution degradation, and the input and output sizes remain the same.</p>
<p>At the <inline-formula>
<mml:math id="M4">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>th stage in <xref ref-type="disp-formula" rid="EQ5">Equation 1</xref>, it is assumed that <inline-formula>
<mml:math id="M5">
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula>(&#x00B7;) represents the transformation function of the <inline-formula>
<mml:math id="M6">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula>th MV2-Block. For example, <inline-formula>
<mml:math id="M7">
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> denotes the result produced by the 4th MN2-Block in the <inline-formula>
<mml:math id="M8">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>th stage. The MobileViT-block module at the <inline-formula>
<mml:math id="M9">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>th stage has a transformation function denoted as <inline-formula>
<mml:math id="M10">
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:math>
</inline-formula>(&#x00B7;). It is important to emphasize that there is a singular MobileViT-block present at each stage.</p>
<p>Moreover, if we represent the output generated by the <inline-formula>
<mml:math id="M11">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula>th MN2-Block module during the <inline-formula>
<mml:math id="M12">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>th stage as (<inline-formula>
<mml:math id="M13">
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula>), it is important to highlight that just the first two MN2-Block components reduce the resolution of the original feature map. Thus, <inline-formula>
<mml:math id="M14">
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> belongs to the set of elements in <inline-formula>
<mml:math id="M15">
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:mfrac>
<mml:mi>H</mml:mi>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mfrac>
<mml:mi>W</mml:mi>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M16">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> belong to the collection <inline-formula>
<mml:math id="M17">
<mml:mfenced open="{" close="}" separators=",,">
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
<mml:mn>3</mml:mn>
</mml:mfenced>
<mml:mtext>,</mml:mtext>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M18">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula> is an element of the set <inline-formula>
<mml:math id="M19">
<mml:mfenced open="{" close="}" separators=",,,">
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
<mml:mn>3</mml:mn>
<mml:mn>4</mml:mn>
</mml:mfenced>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math id="M20">
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> represents the numerical value assigned to the feature channel at the <inline-formula>
<mml:math id="M21">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>th stage.<disp-formula id="E1">
<mml:math id="M22">
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>M</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="E2">
<mml:math id="M23">
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>3</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>3</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>M</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ5">
<label>(1)</label>
<mml:math id="M24">
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>M</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
</p>
<p>The channel attention module is denoted by <inline-formula>
<mml:math id="M25">
<mml:mi>C</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>M</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mo>&#x00B7;</mml:mo>
</mml:mfenced>
</mml:math>
</inline-formula>. It is reasonable to assume that at this stage <inline-formula>
<mml:math id="M26">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>, the output of the left path is <inline-formula>
<mml:math id="M27">
<mml:msup>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> and the output of the right path is <inline-formula>
<mml:math id="M28">
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:math>
</inline-formula>. The formulas <xref ref-type="disp-formula" rid="EQ6">Equation 2</xref> can be used to compute <inline-formula>
<mml:math id="M29">
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:math>
</inline-formula>and <inline-formula>
<mml:math id="M30">
<mml:msup>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:math>
</inline-formula>:<disp-formula id="E3">
<mml:math id="M31">
<mml:msup>
<mml:mi>L</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="normal">Split</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">Concat</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>M</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>3</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ6">
<label>(2)</label>
<mml:math id="M32">
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="normal">Split</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">Concat</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>M</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>3</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
</p>
<p>Here, <inline-formula>
<mml:math id="M33">
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> represents a <inline-formula>
<mml:math id="M34">
<mml:mn>1</mml:mn>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> convolution operation, <inline-formula>
<mml:math id="M35">
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> represents the feature map, and <inline-formula>
<mml:math id="M36">
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>3</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M37">
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> represent the outputs of the third and fourth MN2-Block modules at stage <inline-formula>
<mml:math id="M38">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>, respectively. The <inline-formula>
<mml:math id="M39">
<mml:mi mathvariant="italic">Concat</mml:mi>
</mml:math>
</inline-formula> function concatenates the feature map <inline-formula>
<mml:math id="M40">
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> with the output of the channel attention module applied to the sum of <inline-formula>
<mml:math id="M41">
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>3</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M42">
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula>. The <inline-formula>
<mml:math id="M43">
<mml:mi mathvariant="italic">Split</mml:mi>
</mml:math>
</inline-formula> function splits the resulting tensor into multiple parts.</p>
</sec>
<sec id="sec8">
<label>2.4.2</label>
<title>Decoder sub-network</title>
<p>In the context of a decoder sub-network shown in <xref ref-type="fig" rid="fig1">Figure 1B</xref>, the function <inline-formula>
<mml:math id="M44">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mo>&#x00B7;</mml:mo>
</mml:mfenced>
</mml:math>
</inline-formula> represents the operation of the Feature Fusion Module (FFM) like figured in <xref ref-type="disp-formula" rid="EQ7">Equation 3</xref>. The module takes input <inline-formula>
<mml:math id="M45">
<mml:mi>I</mml:mi>
</mml:math>
</inline-formula> and processes it through a series of transformations involving convolutional operations and batch normalization. The formula provided is as follows:<disp-formula id="EQ7">
<label>(3)</label>
<mml:math id="M46">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mspace width="thickmathspace"/>
<mml:mfenced open="(" close=")">
<mml:mi>I</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="normal">BatchNorm</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi>F</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mi>I</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>F</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mi>I</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
</p>
<p>
<inline-formula>
<mml:math id="M47">
<mml:msup>
<mml:mover accent="true">
<mml:mi>F</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> represents a divided convolution functional with a kernel size of 3&#x202F;&#x00D7;&#x202F;3 and an increase rate of 1, which is equal to a conventional convolution. <inline-formula>
<mml:math id="M48">
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>F</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> represents a dilated convolution operation with a kernel size of 3&#x202F;&#x00D7;&#x202F;3 and an expansion rate of 2. This indicates a convolution with a dilation process using a 3&#x202F;&#x00D7;&#x202F;3 kernel area and an expansion rate that is 1, equivalent to a conventional. BatchNorm refers to the batch normalization operation that standardizes the inputs to a layer for each mini-batch.</p>
<p>In the decoder stage of the network, the feature maps are represented as <inline-formula>
<mml:math id="M49">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:mfrac>
<mml:mi>H</mml:mi>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mfrac>
<mml:mi>W</mml:mi>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M50">
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced open="{" close="}" separators=",,">
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
<mml:mn>3</mml:mn>
</mml:mfenced>
</mml:math>
</inline-formula>. After the encoder phase, the operation on the feature map <inline-formula>
<mml:math id="M51">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> (at the third stage of the decoder) is defined <inline-formula>
<mml:math id="M52">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="italic">BatchNorm</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>3</mml:mn>
<mml:mn>3</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>3</mml:mn>
<mml:mn>4</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>. The two main steps for calculating <inline-formula>
<mml:math id="M53">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mspace width="thickmathspace"/>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">f</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> as described. This process entails increasing the resolution of the data and then combining it with the results from the earlier stage of the encoding process. The initial step involves performing up-sampling and Feature Fusion Mapping (FFM). Up-sampling is intended to adjust the feature size to match the output size of the encoder from the previous stage, as illustrated in <xref ref-type="disp-formula" rid="EQ8">Equation 4</xref>. This process yields an intermediate variable, denoted as <inline-formula>
<mml:math id="M54">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>. The subsequent step entails a feature fusion operation with the encoder&#x2019;s output from the prior stage, as demonstrated in <xref ref-type="disp-formula" rid="EQ9">Equation 5</xref>:<disp-formula id="EQ8">
<label>(4)</label>
<mml:math id="M55">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">Upsample</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>M</mml:mi>
<mml:mfenced open="(" close=")">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ9">
<label>(5)</label>
<mml:math id="M56">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="normal">PReLU</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">BetchNorm</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>3</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>M</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi>i</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
</p>
<p>Upsample(&#x00B7;, t) denotes the procedure of augmenting the data map in accordance with the parameter t by bilinear interpolation. The PReLU function of activation is denoted as PReLU(&#x00B7;), whereas batch normalization is indicated as BatchNorm(&#x00B7;). Once all <inline-formula>
<mml:math id="M57">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mspace width="thickmathspace"/>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> are calculated, a final prediction is obtained through a <inline-formula>
<mml:math id="M58">
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> convolution.<disp-formula id="EQ10">
<label>(6)</label>
<mml:math id="M59">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="normal">Softmax</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">Upsample</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mi>i</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
</p>
<p>Where <inline-formula>
<mml:math id="M60">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> is an element of the set <inline-formula>
<mml:math id="M61">
<mml:mfenced open="{" close="}" separators=",,">
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
<mml:mn>3</mml:mn>
</mml:mfenced>
<mml:mtext>,</mml:mtext>
</mml:math>
</inline-formula> Softmax(&#x00B7;) denotes the function that activates softmax. <inline-formula>
<mml:math id="M62">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> denotes the predicted class label map, with <inline-formula>
<mml:math id="M63">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>being the final output prediction, as demonstrated in <xref ref-type="disp-formula" rid="EQ10">Equation 6</xref>.</p>
</sec>
<sec id="sec9">
<label>2.4.3</label>
<title>MobileViT block</title>
<p>Vision Transformers (ViTs) can achieve comparable accuracy to Convolutional Neural Networks (CNNs) in image identification tasks, especially when trained on extensive datasets (<xref ref-type="bibr" rid="ref9">Dosovitskiy, 2021</xref>). On the other hand, unlike CNNs, ViTs are difficult to optimize and require a large amount of data for training. Research indicates that the suboptimal performance of ViTs is due to a lack of inductive biase (<xref ref-type="bibr" rid="ref24">Lee et al., 2019</xref>; <xref ref-type="bibr" rid="ref32">Petit et al., 2021</xref>; <xref ref-type="bibr" rid="ref50">Zhou et al., 2024</xref>). Inductive biases, while beneficial, also have drawbacks for CNNs; they enable CNNs to capture local spatial information but can limit the network&#x2019;s overall performance.</p>
<p>However, the transformer&#x2019;s self-attention system has the capacity to collect global data. Numerous transformers and CNNs combinations have been investigated to overcome their respective deficiencies. ConViT (<xref ref-type="bibr" rid="ref7">d&#x2019;Ascoli et al., 2022</xref>) uses gated positional self-attention soft convolutional inductive biases. Semantic segmentation models such as ACNET (<xref ref-type="bibr" rid="ref15">Hu et al., 2019</xref>) and CMANet (<xref ref-type="bibr" rid="ref51">Zhu et al., 2022</xref>) have been developed; however, many of these models are computationally intensive. The possibility of leveraging the strengths of both CNNs and ViTs to construct a lightweight network for visual tasks remains an area of ongoing exploration. MobileViT suggests that such an approach is indeed feasible. In this paper, we first examine the calculations involved in MobileViT.</p>
<p>The MobileViT Block, seen in <xref ref-type="fig" rid="fig1">Figure 1D</xref>, shares an identical structure with the MobileViT Unit (<xref ref-type="bibr" rid="ref27">Mehta and Apple, 2022</xref>). The following four phases are applied to a given source tensor <inline-formula>
<mml:math id="M64">
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>:<list list-type="bullet">
<list-item>
<p>The input tensor &#x1d44b; is first passed through an <inline-formula>
<mml:math id="M65">
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula> standard convolution layer, followed by a <inline-formula>
<mml:math id="M66">
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> convolution layer to generate <inline-formula>
<mml:math id="M67">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. The <inline-formula>
<mml:math id="M68">
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula> convolution stage captures and represents nearby spatial details, whereas the <inline-formula>
<mml:math id="M69">
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> convolution transforms the tensors into higher-dimensional spaces (with <inline-formula>
<mml:math id="M70">
<mml:mi>d</mml:mi>
</mml:math>
</inline-formula>dimensions, where <inline-formula>
<mml:math id="M71">
<mml:mi>d</mml:mi>
</mml:math>
</inline-formula> is more than <inline-formula>
<mml:math id="M72">
<mml:mi>c</mml:mi>
</mml:math>
</inline-formula>) by acquiring knowledge of a linear combination of the input channels.</p>
</list-item>
<list-item>
<p>In order to incorporate spatial inductive bias into MobileViT&#x2019;s learning process, the input <inline-formula>
<mml:math id="M73">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is divided into <inline-formula>
<mml:math id="M74">
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula> non-overlapping flattening patches <inline-formula>
<mml:math id="M75">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>U</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:msup>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M76">
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>h</mml:mi>
<mml:mi>w</mml:mi>
</mml:math>
</inline-formula>. The total number of patches is represented by the formula <inline-formula>
<mml:math id="M77">
<mml:mi>N</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:mfrac>
</mml:math>
</inline-formula>, where h and w are the physical dimensions of every single patch, individually.</p>
</list-item>
<list-item>
<p>The transformer is then applied to encode the relationships between the patches through the following operation, as demonstrated in <xref ref-type="disp-formula" rid="EQ11">Equation 7</xref>
</p>
</list-item>
</list>
<disp-formula id="EQ11">
<label>(7)</label>
<mml:math id="M78">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>p</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="normal">Transformer</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>U</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>p</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>P</mml:mi>
</mml:math>
</disp-formula>
<list list-type="bullet">
<list-item>
<p>The resulting <inline-formula>
<mml:math id="M79">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> is then folded back to obtain <inline-formula>
<mml:math id="M80">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>F</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>Ultimately, <inline-formula>
<mml:math id="M81">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>F</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is transformed into a space with fewer dimensions (<inline-formula>
<mml:math id="M82">
<mml:mi>C</mml:mi>
</mml:math>
</inline-formula> dimensions) using point-by-point convolution and then merging with <inline-formula>
<mml:math id="M83">
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula> using concatenation.</p>
</list-item>
</list>
</p>
<p>The second phase contains the algorithm&#x2019;s core. The input image <inline-formula>
<mml:math id="M84">
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula>, which has dimensions <inline-formula>
<mml:math id="M85">
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:math>
</inline-formula>, is separated into patches in the standard ViT structure. Subsequently, every patch undergoes a linear transformation to convert it into a vector. These vectors are then encoded with positional information. Furthermore, the interconnections between the patches are acquired by employing <inline-formula>
<mml:math id="M86">
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula> transformer blocks.</p>
<p>Contrary to ViT, the MobileViT algorithm preserves both the patch order and the physical order of pixels inside each of the patches during its second stage. It is crucial to emphasize that the values of <inline-formula>
<mml:math id="M87">
<mml:mi>w</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M88">
<mml:mi>h</mml:mi>
</mml:math>
</inline-formula> must be exact divisors of <inline-formula>
<mml:math id="M89">
<mml:mi>W</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M90">
<mml:mi>H</mml:mi>
</mml:math>
</inline-formula>, respectively.</p>
<p>Local information can be encoded by the relationship <inline-formula>
<mml:math id="M91">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>U</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>. The yellow pixels inside a patch have the ability to aggregate data from the pixels that surround them in that patch, as seen in <xref ref-type="fig" rid="fig1">Figure 1C</xref>. <inline-formula>
<mml:math id="M92">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>G</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> accomplishes the worldwide data encoding of the transformer by encoding inter-patch connections at the <inline-formula>
<mml:math id="M93">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula>-th place of every patch. The red pixel keeps track of every one of the pixels that encode the full image since, as <xref ref-type="fig" rid="fig1">Figure 1C</xref> illustrates, it can identify the yellow pixel that is at the same location in other patches. The lightweight aspect of the model is enhanced by the dot product procedure, which selects only pixels that are in the same position.</p>
<p>According to <xref ref-type="bibr" rid="ref27">Mehta and Apple (2022)</xref>, ordinary convolution can be broken down into three steps: unfolding, matrix multiplication, and folding. Based on the previously indicated computation, a layer of convolution and the Unfold operation carry out the local feature modeling, providing them with convolution-like inductive biases. Next, global feature modeling is carried out using the Transformer &#x2192; Fold sequence, which gives the MobileViT block global processing power.</p>
</sec>
</sec>
<sec id="sec10">
<label>2.5</label>
<title>Multi query attention</title>
<p>
<xref ref-type="fig" rid="fig2">Figure 2A</xref> displays the configuration of a standard non-local block (<xref ref-type="bibr" rid="ref43">Wang et al., 2018</xref>). The non-local block (<xref ref-type="bibr" rid="ref43">Wang et al., 2018</xref>) first requires the computation of the similarity between all places. This is accomplished by performing matrix multiplication on an input <inline-formula>
<mml:math id="M94">
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. The primary computational procedure in the non-local block can be succinctly described as consisting of the following five steps:<list list-type="bullet">
<list-item>
<p>The source feature <inline-formula>
<mml:math id="M95">
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula> is subjected to three <inline-formula>
<mml:math id="M96">
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> convolutions, denoted as <inline-formula>
<mml:math id="M97">
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>&#x03D5;</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M98">
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math id="M99">
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>&#x03B3;</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>, resulting in the transformation of <inline-formula>
<mml:math id="M100">
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula> into <inline-formula>
<mml:math id="M101">
<mml:mi>&#x03D5;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M102">
<mml:mi>&#x03B8;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mtext>,</mml:mtext>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M103">
<mml:mi>&#x03B3;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. The three numbers correspond to the query, key, and value, respectively. They are used to change the total amount of streams from &#x1d436; to <inline-formula>
<mml:math id="M104">
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>A similarity matrix M is created by flattening the query, key, and value to size <inline-formula>
<mml:math id="M105">
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M106">
<mml:mi>N</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>H</mml:mi>
<mml:mi>W</mml:mi>
</mml:math>
</inline-formula>. The matrix <inline-formula>
<mml:math id="M107">
<mml:mi>M</mml:mi>
</mml:math>
</inline-formula> is calculated to determine the similarity, as demonstrated in <xref ref-type="disp-formula" rid="EQ12">Equation 8</xref>:</p>
</list-item>
</list>
<disp-formula id="EQ12">
<label>(8)</label>
<mml:math id="M108">
<mml:mi>M</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>&#x03D5;</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>&#x03B8;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</disp-formula>
<list list-type="bullet">
<list-item>
<p>The matrix <inline-formula>
<mml:math id="M109">
<mml:mi>M</mml:mi>
</mml:math>
</inline-formula> is normalized using a normalization function such as softmax: <inline-formula>
<mml:math id="M110">
<mml:mover accent="true">
<mml:mi>M</mml:mi>
<mml:mo stretchy="true">&#x2192;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="normal">softmax</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mtext>.</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>The matrix of attention <inline-formula>
<mml:math id="M111">
<mml:mi>A</mml:mi>
</mml:math>
</inline-formula> is then derived, as demonstrated in <xref ref-type="disp-formula" rid="EQ13">Equation 9</xref>:</p>
</list-item>
</list>
<disp-formula id="EQ13">
<label>(9)</label>
<mml:math id="M112">
<mml:mi>A</mml:mi>
<mml:mo>=</mml:mo>
<mml:mover accent="true">
<mml:mi>M</mml:mi>
<mml:mo stretchy="true">&#x2192;</mml:mo>
</mml:mover>
<mml:mo>&#x00D7;</mml:mo>
<mml:msup>
<mml:mi>&#x03B3;</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:msup>
</mml:math>
</disp-formula>
<list list-type="bullet">
<list-item>
<p>The final result is computed as: <inline-formula>
<mml:math id="M113">
<mml:mi>Y</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula>. The channel dimension is adjusted from <inline-formula>
<mml:math id="M114">
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>back to <inline-formula>
<mml:math id="M115">
<mml:mi>C</mml:mi>
</mml:math>
</inline-formula> by a <inline-formula>
<mml:math id="M116">
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> convolution, denoted as <inline-formula>
<mml:math id="M117">
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>.</p>
</list-item>
</list>
</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Architecture of a standard <bold>(A)</bold> non-local block, <bold>(B)</bold> the asymmetric non-local block, and <bold>(C)</bold> Multi Query attention module.</p>
</caption>
<graphic xlink:href="fcomp-07-1510252-g002.tif"/>
</fig>
<p>Reassessing the Non-local Asymmetric Block</p>
<p>The computational complexity of the global attention block can be described as <inline-formula>
<mml:math id="M118">
<mml:mi>O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:msup>
<mml:mi>N</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mi>O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:msup>
<mml:mi>H</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>. The calculation efficiency is mostly affected by N&#x2019;s size. To fix this, reduce <inline-formula>
<mml:math id="M119">
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula> to <inline-formula>
<mml:math id="M120">
<mml:mi>S</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x226A;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> without altering output size. <xref ref-type="bibr" rid="ref52">Zhu et al. (2019)</xref> offers the asymmetric non-local block, whose construction is shown in <xref ref-type="fig" rid="fig2">Figure 2B</xref>, to tackle this problem.</p>
<p>The asymmetrical pyramid non-local block (APNB) is a modified version of this block that incorporates pyramid pooling within the non-local block in order to decrease computational expenses. One more thing is added after <inline-formula>
<mml:math id="M121">
<mml:mi>&#x03B8;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M122">
<mml:mi>&#x03B3;</mml:mi>
</mml:math>
</inline-formula>: a spatial pyramid pooling function (<xref ref-type="bibr" rid="ref23">Lazebnik et al., 2006</xref>) to pick out a few good anchor points. When the spatial pyramid pooling modules are <inline-formula>
<mml:math id="M123">
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M124">
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>&#x03B3;</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula>, with n denoting the pooling layer&#x2019;s output size (width or height) and <inline-formula>
<mml:math id="M125">
<mml:mi>n</mml:mi>
<mml:munder accentunder="true">
<mml:mo>&#x2282;</mml:mo>
<mml:mo>_</mml:mo>
</mml:munder>
<mml:mfenced open="{" close="}" separators=",,,">
<mml:mn>1</mml:mn>
<mml:mn>3</mml:mn>
<mml:mn>6</mml:mn>
<mml:mn>8</mml:mn>
</mml:mfenced>
</mml:math>
</inline-formula> as per <xref ref-type="bibr" rid="ref52">Zhu et al. (2019)</xref>, the overall amount of sampling anchor points is <inline-formula>
<mml:math id="M126">
<mml:mi>S</mml:mi>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced open="{" close="}" separators=",,,">
<mml:mn>1</mml:mn>
<mml:mn>3</mml:mn>
<mml:mn>6</mml:mn>
<mml:mn>8</mml:mn>
</mml:mfenced>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mi>n</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mn>110</mml:mn>
</mml:math>
</inline-formula>. If we assume that <inline-formula>
<mml:math id="M127">
<mml:mi>H</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>224</mml:mn>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M128">
<mml:mi>W</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>256</mml:mn>
</mml:math>
</inline-formula>, the number of calculations will be reduced by an amount <inline-formula>
<mml:math id="M129">
<mml:mfrac>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mi>S</mml:mi>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>256</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
<mml:mn>110</mml:mn>
</mml:mfrac>
<mml:mo>&#x2248;</mml:mo>
<mml:mn>595</mml:mn>
</mml:math>
</inline-formula>. This adjustment efficiently decreases the value of <inline-formula>
<mml:math id="M130">
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula> to a lower value, <inline-formula>
<mml:math id="M131">
<mml:mi>S</mml:mi>
</mml:math>
</inline-formula>, by selectively sampling a few sample data from <inline-formula>
<mml:math id="M132">
<mml:mi>&#x03B8;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M133">
<mml:mi>&#x03B3;</mml:mi>
</mml:math>
</inline-formula>, instead of utilizing all of the points.</p>
<p>The primary computational procedure in the asymmetrical non-local block entails the subsequent modifications, as demonstrated in <xref ref-type="disp-formula" rid="EQ14 EQ15">Equations 10, 11</xref>:<disp-formula id="EQ14">
<label>(10)</label>
<mml:math id="M134">
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:munder>
<mml:mrow>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="true">&#xFE38;</mml:mo>
</mml:munder>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>E</mml:mi>
<mml:mi>q</mml:mi>
<mml:mo>.</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mn>8</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mo>&#x2192;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:munder>
<mml:mrow>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="true">&#xFE38;</mml:mo>
</mml:munder>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>E</mml:mi>
<mml:mi>q</mml:mi>
<mml:mo>.</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mn>9</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mo>&#x2192;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<disp-formula id="EQ15">
<label>(11)</label>
<mml:math id="M135">
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:munder>
<mml:mrow>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="true">&#xFE38;</mml:mo>
</mml:munder>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>E</mml:mi>
<mml:mi>q</mml:mi>
<mml:mo>.</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mn>12</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mo>&#x2192;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:munder>
<mml:mrow>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="true">&#xFE38;</mml:mo>
</mml:munder>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mi>E</mml:mi>
<mml:mi>q</mml:mi>
<mml:mo>.</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mn>13</mml:mn>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mo>&#x2192;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
</p>
<p>The computational technique for the APNB module may be delineated as follows:</p>
<p>Introduce sampling modules <inline-formula>
<mml:math id="M136">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M137">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>&#x03B3;</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> after <inline-formula>
<mml:math id="M138">
<mml:mi>&#x03B8;</mml:mi>
</mml:math>
</inline-formula> and <italic>&#x03B3;</italic>, accordingly, to sample multiple sparse anchor points. These anchor points are designated as <inline-formula>
<mml:math id="M139">
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M140">
<mml:msub>
<mml:mi>&#x03B3;</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>C</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, respectively.</p>
<p>Generating a similarity matrix <inline-formula>
<mml:math id="M141">
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>, as demonstrated in <xref ref-type="disp-formula" rid="EQ16">Equation 12</xref>:<disp-formula id="EQ16">
<label>(12)</label>
<mml:math id="M142">
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>&#x03D5;</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</disp-formula>
</p>
<p>
<inline-formula>
<mml:math id="M143">
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is normalized. <inline-formula>
<mml:math id="M144">
<mml:msub>
<mml:mover accent="true">
<mml:mi>M</mml:mi>
<mml:mo stretchy="true">&#x2192;</mml:mo>
</mml:mover>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:mfenced>
</mml:math>
</inline-formula>
</p>
<p>
<inline-formula>
<mml:math id="M145">
<mml:mi>f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mo>&#x00B7;</mml:mo>
</mml:mfenced>
</mml:math>
</inline-formula> stands for normalization function.</p>
<p>The attention matrix AP is subsequently computed, as demonstrated in <xref ref-type="disp-formula" rid="EQ17">Equation 13</xref>:<disp-formula id="EQ17">
<label>(13)</label>
<mml:math id="M146">
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>M</mml:mi>
<mml:mo stretchy="true">&#x2192;</mml:mo>
</mml:mover>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msubsup>
<mml:mi>&#x03B3;</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mspace width="1.25em"/>
</mml:mrow>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:msup>
</mml:math>
</disp-formula>
</p>
<p>The final result is obtained by adding the product of the variables <inline-formula>
<mml:math id="M147">
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M148">
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mi>T</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> to the variable X, and assigning it to the variable <inline-formula>
<mml:math id="M149">
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>.</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is a one-by-one convolution. <inline-formula>
<mml:math id="M150">
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>.</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> represents a <inline-formula>
<mml:math id="M151">
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula>convolutions.</p>
</sec>
<sec id="sec11">
<label>2.6</label>
<title>Motivation</title>
<p>The APNB discussed earlier operates with a single data stream as input, while the asymmetrical fusion nonlocal block typically utilizes two data sources: the top-level feature map and the lower-level feature map. In contrast, the proposed MQA mechanism extends this approach by incorporating four input sources. As shown in <xref ref-type="fig" rid="fig1">Figure 1A</xref>, MQA integrates the fundamental value (<inline-formula>
<mml:math id="M152">
<mml:mi>P</mml:mi>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula>), the elevated characteristics <inline-formula>
<mml:math id="M153">
<mml:mi>P</mml:mi>
<mml:mn>2</mml:mn>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M154">
<mml:mi>P</mml:mi>
<mml:mn>3</mml:mn>
</mml:math>
</inline-formula>, and the edge content (<inline-formula>
<mml:math id="M155">
<mml:mi>P</mml:mi>
<mml:mi>e</mml:mi>
</mml:math>
</inline-formula>), allowing for the explicit acquisition of multiple levels of feature representation. By incorporating edge information throughout the semantic segmentation process, the model imposes valuable constraints, enhancing segmentation precision. The cross-entropy loss function further refines this process by measuring the difference between the ground truth (<inline-formula>
<mml:math id="M156">
<mml:mi>G</mml:mi>
<mml:mi>T</mml:mi>
</mml:math>
</inline-formula>) and the feature aggregation map <inline-formula>
<mml:math id="M157">
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>f</mml:mi>
</mml:math>
</inline-formula>), ensuring robust alignment of predictions with the actual data.</p>
<p>Commence the acquisition of a parameter map of features <inline-formula>
<mml:math id="M158">
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mtext>,</mml:mtext>
</mml:math>
</inline-formula> and additional feature maps <inline-formula>
<mml:math id="M159">
<mml:mi>X</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M160">
<mml:mi>X</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math id="M161">
<mml:mi>X</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. Five 1&#x202F;&#x00D7;&#x202F;1 convolutions, denoted as <inline-formula>
<mml:math id="M162">
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>&#x03D5;</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>&#x03B3;</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mtext>,</mml:mtext>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M163">
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, are applied to transform these input maps into new feature maps: <inline-formula>
<mml:math id="M164">
<mml:mi>&#x03D5;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>&#x03B3;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mtext>,</mml:mtext>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M165">
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, as demonstrated in <xref ref-type="disp-formula" rid="EQ18">Equation 14</xref>:<disp-formula id="E4">
<mml:math id="M166">
<mml:mi>&#x03D5;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>W</mml:mi>
<mml:mi>&#x03D5;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>X</mml:mi>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>&#x03B8;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>W</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>X</mml:mi>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>&#x03B3;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>W</mml:mi>
<mml:mi>&#x03B3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mi>X</mml:mi>
</mml:mfenced>
<mml:mtext>,</mml:mtext>
</mml:math>
</disp-formula>
<disp-formula id="EQ18">
<label>(14)</label>
<mml:math id="M167">
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
</p>
<p>The parameters <inline-formula>
<mml:math id="M168">
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M169">
<mml:mi>X</mml:mi>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M170">
<mml:mi>X</mml:mi>
<mml:mn>2</mml:mn>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math id="M171">
<mml:mi>X</mml:mi>
<mml:mn>3</mml:mn>
</mml:math>
</inline-formula> in this experiment correspond to the results <inline-formula>
<mml:math id="M172">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M173">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M174">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>, and P3 from MoNetViT. The sample as well as main computation methodologies within the MQA module were equivalent to those in APNB. The selection units <inline-formula>
<mml:math id="M175">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>&#x03B3;</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>&#x03D5;</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math id="M176">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> are utilized to sample multiple sparse anchor points. These anchor points are represented as <inline-formula>
<mml:math id="M177">
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x03B3;</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x03D5;</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>&#x03D5;</mml:mi>
<mml:msub>
<mml:mn>1</mml:mn>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M178">
<mml:mi>&#x03D5;</mml:mi>
<mml:msub>
<mml:mn>2</mml:mn>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M179">
<mml:mi>S</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>1</mml:mn>
<mml:mtext>,</mml:mtext>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M180">
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:math>
</inline-formula> denote the number of sampled anchor points. <inline-formula>
<mml:math id="M181">
<mml:mi>S</mml:mi>
</mml:math>
</inline-formula> is less than <inline-formula>
<mml:math id="M182">
<mml:mi>S</mml:mi>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula>, which is less than <inline-formula>
<mml:math id="M183">
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math id="M184">
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:math>
</inline-formula> is much less than <inline-formula>
<mml:math id="M185">
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula>. Mathematically, this is computed using the following <xref ref-type="disp-formula" rid="EQ19">Equation 15</xref>:<disp-formula id="E5">
<mml:math id="M186">
<mml:mi>K</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>&#x03B8;</mml:mi>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>&#x03B3;</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>&#x03B3;</mml:mi>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>&#x03D5;</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mi>&#x03D5;</mml:mi>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ19">
<label>(15)</label>
<mml:math id="M187">
<mml:mi>Q</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mspace width="thickmathspace"/>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
</p>
<p>The correlation matrix <inline-formula>
<mml:math id="M188">
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> for <inline-formula>
<mml:math id="M189">
<mml:mi>Q</mml:mi>
</mml:math>
</inline-formula> and anchoring <inline-formula>
<mml:math id="M190">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> is displayed here, as demonstrated in <xref ref-type="disp-formula" rid="E6">Equation 16</xref>:<disp-formula id="E6">
<label>(16)</label>
<mml:math id="M191">
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>K</mml:mi>
</mml:math>
</disp-formula>
</p>
<p>The dimensions of the <inline-formula>
<mml:math id="M192">
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>are <inline-formula>
<mml:math id="M193">
<mml:mi>S</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M194">
<mml:mi>S</mml:mi>
</mml:math>
</inline-formula> is significantly smaller than <inline-formula>
<mml:math id="M195">
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula>. Next, the process of normalization is carried out on <inline-formula>
<mml:math id="M196">
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>, which enables the calculation of <inline-formula>
<mml:math id="M197">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>.</p>
<p>The ultimate result of the initial layer is, as demonstrated in <xref ref-type="disp-formula" rid="EQ20">Equation 17</xref>:<disp-formula id="EQ20">
<label>(17)</label>
<mml:math id="M198">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</disp-formula>
</p>
<p>The value of <inline-formula>
<mml:math id="M199">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> is equal to the product of <inline-formula>
<mml:math id="M200">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M201">
<mml:mi>V</mml:mi>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math id="M202">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> belongs to the set of <inline-formula>
<mml:math id="M203">
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. The similarity matrix for levels 2, 3, and 4 is computed using an analogy, as demonstrated in <xref ref-type="disp-formula" rid="EQ21">Equation 18</xref>:<disp-formula id="E7">
<mml:math id="M204">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>Q</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</disp-formula>
<disp-formula id="E8">
<mml:math id="M205">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>Q</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</disp-formula>
<disp-formula id="EQ21">
<label>(18)</label>
<mml:math id="M206">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>Q</mml:mi>
<mml:mn>3</mml:mn>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</disp-formula>
</p>
<p>The equation <inline-formula>
<mml:math id="M207">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> is equal to the product of <inline-formula>
<mml:math id="M208">
<mml:msubsup>
<mml:mi>Q</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M209">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>. The equation <inline-formula>
<mml:math id="M210">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> is equal to the product of <inline-formula>
<mml:math id="M211">
<mml:msubsup>
<mml:mi>Q</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M212">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>. The equation <inline-formula>
<mml:math id="M213">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> is equal to the product of <inline-formula>
<mml:math id="M214">
<mml:msubsup>
<mml:mi>Q</mml:mi>
<mml:mn>3</mml:mn>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M215">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>.</p>
<p>The ultimate result of levels 2, 3, and 4 is determined in the following manner <xref ref-type="disp-formula" rid="EQ22">Equation 19</xref>:<disp-formula id="E9">
<mml:math id="M216">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</disp-formula>
<disp-formula id="E10">
<mml:math id="M217">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</disp-formula>
<disp-formula id="EQ22">
<label>(19)</label>
<mml:math id="M218">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x00D7;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</disp-formula>
</p>
<p>The value of <inline-formula>
<mml:math id="M219">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> is equal to the product of <inline-formula>
<mml:math id="M220">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M221">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M222">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> belongs to the set <inline-formula>
<mml:math id="M223">
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. The value of <inline-formula>
<mml:math id="M224">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> is equal to the product of <inline-formula>
<mml:math id="M225">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M226">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M227">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> belongs to the set <inline-formula>
<mml:math id="M228">
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. The value of <inline-formula>
<mml:math id="M229">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> is equal to the product of <inline-formula>
<mml:math id="M230">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M231">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M232">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> belongs to the set <inline-formula>
<mml:math id="M233">
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>.</p>
<p>The final result is represented as <inline-formula>
<mml:math id="M234">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>K</mml:mi>
<mml:mi>V</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> belonging to the set of <inline-formula>
<mml:math id="M235">
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. The temporal complexity can be represented as <inline-formula>
<mml:math id="M236">
<mml:mi>O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>3</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, which is significantly lower than <inline-formula>
<mml:math id="M237">
<mml:mi>O</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:msup>
<mml:mi>N</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> in the conventional non-local block.</p>
</sec>
<sec id="sec12">
<label>2.7</label>
<title>Loss function</title>
<p>The function that measures loss is defined like the ones used by <xref ref-type="bibr" rid="ref47">Yeung et al. (2022)</xref>. The loss function comprises two components: <inline-formula>
<mml:math id="M238">
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi mathvariant="italic">edge</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M239">
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi mathvariant="italic">seg</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>. <xref ref-type="disp-formula" rid="EQ23 EQ24">Equations 20, 21</xref> (<xref ref-type="bibr" rid="ref47">Yeung et al., 2022</xref>) display the loss function.<disp-formula id="EQ23">
<label>(20)</label>
<mml:math id="M240">
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi mathvariant="italic">edge</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>log</mml:mo>
<mml:mfenced open="(" close=")">
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>log</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
</p>
<p>where ground-truth (GT) and the anticipated edge map Pe&#x2019;s coordinates for each pixel point are represented by (<inline-formula>
<mml:math id="M241">
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>).</p>
<p>IoU loss and a conventional cross-entropy loss make up the two components of the <inline-formula>
<mml:math id="M242">
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi mathvariant="italic">seg</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> loss function.<disp-formula id="EQ24">
<label>(21)</label>
<mml:math id="M243">
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi mathvariant="normal">seg</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msubsup>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
</mml:mrow>
<mml:mi mathvariant="normal">w</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msubsup>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">I</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">U</mml:mi>
</mml:mrow>
<mml:mi mathvariant="normal">w</mml:mi>
</mml:msubsup>
</mml:math>
</disp-formula>
</p>
</sec>
<sec id="sec13">
<label>2.8</label>
<title>Experiment setup</title>
<p>The NVIDIA Tesla T4 GPU was used to train the model within the PyTorch framework for this project. The model underwent training for 50 epochs using a batch size of 16, with the Adam optimizer and a starting learning rate of 1e-3. A learning rate reduction factor of 0.5 was applied when no improvement was observed for 5 consecutive epochs (patience set to 5). The specific hyperparameters utilized in this investigation are outlined in <xref ref-type="table" rid="tab1">Table 1</xref>. In addition, the experiment aimed to compare the proposed MoNetViT model with several state-of-the-art methods, including TransFuse (<xref ref-type="bibr" rid="ref49">Zhang et al., 2021</xref>), Inf-Net (<xref ref-type="bibr" rid="ref12">Fan et al., 2020</xref>), U-Net (<xref ref-type="bibr" rid="ref34">Ronneberger et al., 2015</xref>), U-Net++ (<xref ref-type="bibr" rid="ref22">Kwak and Sung, 2021</xref>), Mini-Seg (<xref ref-type="bibr" rid="ref19">Kim et al., 2023</xref>), and DeepLabV3+ (<xref ref-type="bibr" rid="ref2">Asadi Shamsabadi et al., 2022</xref>), for multi-class segmentation in ArUco marker identification.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Network hyperparameter.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" char="&#x00D7;">Hyperparameter</th>
<th align="char" valign="top" char="&#x00D7;">Options</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Resize the images</td>
<td align="center" valign="middle">224&#x00D7;224</td>
</tr>
<tr>
<td align="left" valign="middle">Epochs</td>
<td align="center" valign="middle">50</td>
</tr>
<tr>
<td align="left" valign="middle">Batch size</td>
<td align="center" valign="middle">16</td>
</tr>
<tr>
<td align="left" valign="middle">Optimizer</td>
<td align="center" valign="middle">Adam</td>
</tr>
<tr>
<td align="left" valign="middle">Learning rate (Lr)</td>
<td align="center" valign="middle">1e &#x2013; 3</td>
</tr>
<tr>
<td align="left" valign="middle">Factor</td>
<td align="center" valign="middle">0.5</td>
</tr>
<tr>
<td align="left" valign="middle">Patience</td>
<td align="center" valign="middle">5</td>
</tr>
<tr>
<td align="left" valign="middle">
<inline-formula>
<mml:math id="M244">
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M245">
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>
</td>
<td align="center" valign="middle">0.2, 0.8</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The researchers conducted experiments using freely available datasets for ArUco manual labeling. The dataset employed in this study is an open-source resource that has been fully labeled to indicate various classes of ArUco markers, making it ideal for training models in identification and classification tasks. The dataset preparation involved several pre-processing steps to enhance model generalizability and ensure reproducibility. Images were resized to a resolution of 224&#x202F;&#x00D7;&#x202F;224, and pixel values were normalized to the range 0 and 1. Additionally, data augmentation techniques such as random rotation, flipping, and contrast adjustment were applied to improve the model&#x2019;s ability to generalize across varied conditions. The complete dataset, including all necessary images and labels for training and testing models in ArUco marker identification and classification tasks, can be accessed and downloaded from the following link: <ext-link xlink:href="https://universe.roboflow.com/loliktry/dataarucomustofa/dataset/5" ext-link-type="uri">https://universe.roboflow.com/loliktry/dataarucomustofa/dataset/5</ext-link>. It was specifically utilized in the multi-class segmentation experiments. <xref ref-type="table" rid="tab2">Table 2</xref> provides a comprehensive overview of the specific details of this dataset.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Specifications of datasets.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="center" valign="top" char="&#x00D7;">Marker class</th>
<th align="char" valign="top" char="&#x00D7;">No. training</th>
<th align="char" valign="top" char="&#x00D7;">No. valid.</th>
<th align="char" valign="top" char="&#x00D7;">No. testing</th>
<th align="char" valign="top" char="&#x00D7;">Total images</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">1</td>
<td align="center" valign="top">853</td>
<td align="center" valign="top">188</td>
<td align="center" valign="top">289</td>
<td align="center" valign="top">1,330</td>
</tr>
<tr>
<td align="left" valign="top">2</td>
<td align="center" valign="top">904</td>
<td align="center" valign="top">238</td>
<td align="center" valign="top">269</td>
<td align="center" valign="top">1,411</td>
</tr>
<tr>
<td align="left" valign="top">3</td>
<td align="center" valign="top">895</td>
<td align="center" valign="top">237</td>
<td align="center" valign="top">271</td>
<td align="center" valign="top">1,403</td>
</tr>
<tr>
<td align="left" valign="top">Total</td>
<td align="center" valign="top">2,652</td>
<td align="center" valign="top">663</td>
<td align="center" valign="top">829</td>
<td align="center" valign="top">4,144</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Furthermore, the present study utilized identical methodologies as <xref ref-type="bibr" rid="ref12">Fan et al. (2020)</xref> to assess the performance of the model. Standard criteria, such as accuracy, specificity, sensitivity, and Dice similarity coefficient, comprise the assessment metrics. In addition, it uses several metrics from object recognition evaluation methods, such as the design measure, the enhanced alignment value (<xref ref-type="bibr" rid="ref13">Fan et al., 2018</xref>), and the mean absolute error.</p>
</sec>
</sec>
<sec sec-type="results" id="sec14">
<label>3</label>
<title>Results</title>
<p>The dataset utilized in this study is notably large-scale, comprising a total of 4,144 slices. Consequently, the following sections will focus exclusively on a detailed analysis of the results, offering an in-depth examination of the findings and their implications.</p>
<sec id="sec15">
<label>3.1</label>
<title>Three-class ArUco marker labeling results</title>
<p>The segmentation data findings for ArUco Marker on the dataset are displayed in <xref ref-type="fig" rid="fig3">Figure 3</xref>, demonstrating that MoNetViT in this study exhibits superior performance compared to other baseline models. U-Net and U-Net++ have low Dice scores and sensitivities, resulting in large unsegmented areas. Inf-Net and Mini-Seg show slight improvements but still lack accurate boundary detection. While TransFuse, a CNN&#x202F;+&#x202F;Transformer model, was evaluated, it is not included in <xref ref-type="fig" rid="fig3">Figure 3</xref> due to its very low Dice score. DeepLabV3+ performs reasonably well but falls short of MoNetViT. Overall, MoNetViT achieves the highest Dice, sensitivity, and specificity scores, along with the lowest MAE, indicating its superior segmentation precision.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Comparison of segmentation results of three-class labeling.</p>
</caption>
<graphic xlink:href="fcomp-07-1510252-g003.tif"/>
</fig>
<p>The MoNetViT model shows superior performance over other state-of-the-art models, including U-Net, U-Net++, Mini-Seg, Inf-Net, TransFuse, and DeepLabV3+, across key evaluation metrics. As summarized in the <xref ref-type="table" rid="tab3">Table 3</xref>, MoNetViT achieves the highest Dice score (0.9584), sensitivity (0.9424), specificity (0.9424), structural similarity <inline-formula>
<mml:math id="M246">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>&#x03B1;</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> 0.9923), and mean edge accuracy <inline-formula>
<mml:math id="M247">
<mml:msubsup>
<mml:mi>E</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi mathvariant="italic">mean</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> (0.9381), while also attaining the lowest Mean Absolute Error (MAE) of 0.0077.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Result of three-class labeling.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="middle" char="&#x00D7;">Methods</th>
<th align="char" valign="middle" char="&#x00D7;">Param.(M)</th>
<th align="char" valign="middle" char="&#x00D7;">Size(Mb)</th>
<th align="char" valign="middle" char="&#x00D7;">Dice</th>
<th align="char" valign="middle" char="&#x00D7;">Sen.</th>
<th align="char" valign="middle" char="&#x00D7;">Spec.</th>
<th align="char" valign="middle" char="&#x00D7;">
<inline-formula>
<mml:math id="M248">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>&#x03B1;</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>
</th>
<th align="char" valign="middle" char="&#x00D7;">
<inline-formula>
<mml:math id="M249">
<mml:msubsup>
<mml:mi>E</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi mathvariant="italic">mean</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula>
</th>
<th align="char" valign="middle" char="&#x00D7;">MAE</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">U-Net</td>
<td align="center" valign="top">1.953</td>
<td align="center" valign="top">7.438</td>
<td align="center" valign="top">0.5622</td>
<td align="center" valign="top">0.6469</td>
<td align="center" valign="top">0.9803</td>
<td align="center" valign="top">0.9676</td>
<td align="center" valign="top">0.5622</td>
<td align="center" valign="top">0.0344</td>
</tr>
<tr>
<td align="left" valign="top">U-Net++</td>
<td align="center" valign="top">7.783</td>
<td align="center" valign="top">29.69</td>
<td align="center" valign="top">0.6082</td>
<td align="center" valign="top">0.5744</td>
<td align="center" valign="top">0.9013</td>
<td align="center" valign="top">0.9816</td>
<td align="center" valign="top">0.6082</td>
<td align="center" valign="top">0.0351</td>
</tr>
<tr>
<td align="left" valign="top">Inf-Net</td>
<td align="center" valign="top">0.076</td>
<td align="center" valign="top">0.291</td>
<td align="center" valign="top">0.5889</td>
<td align="center" valign="top">0.5585</td>
<td align="center" valign="top">0.8865</td>
<td align="center" valign="top">0.9762</td>
<td align="center" valign="top">0.5889</td>
<td align="center" valign="top">0.0458</td>
</tr>
<tr>
<td align="left" valign="top">Mini-Seg</td>
<td align="center" valign="top">0.038</td>
<td align="center" valign="top">0.145</td>
<td align="center" valign="top">0.6238</td>
<td align="center" valign="top">0.6268</td>
<td align="center" valign="top">0.9537</td>
<td align="center" valign="top">0.9845</td>
<td align="center" valign="top">0.6238</td>
<td align="center" valign="top">0.0292</td>
</tr>
<tr>
<td align="left" valign="top">TransFuse</td>
<td align="center" valign="top">
<bold>0.019</bold>
</td>
<td align="center" valign="top">
<bold>0.073</bold>
</td>
<td align="center" valign="top">0.3231</td>
<td align="center" valign="top">0.3333</td>
<td align="center" valign="top">0.6667</td>
<td align="center" valign="top">0.9403</td>
<td align="center" valign="top">0.3231</td>
<td align="center" valign="top">0.1177</td>
</tr>
<tr>
<td align="left" valign="top">DeepLabV3+</td>
<td align="center" valign="top">13.324</td>
<td align="center" valign="top">50.508</td>
<td align="center" valign="top">0.6351</td>
<td align="center" valign="top">0.6392</td>
<td align="center" valign="top">0.9655</td>
<td align="center" valign="top">0.9881</td>
<td align="center" valign="top">0.6351</td>
<td align="center" valign="top">0.0221</td>
</tr>
<tr>
<td align="left" valign="top">MoNetVIT(ours)</td>
<td align="center" valign="top">1.014</td>
<td align="center" valign="top">3.869</td>
<td align="center" valign="top">
<bold>0.9584</bold>
</td>
<td align="center" valign="top">
<bold>0.9424</bold>
</td>
<td align="center" valign="top">
<bold>0.9424</bold>
</td>
<td align="center" valign="top">
<bold>0.9923</bold>
</td>
<td align="center" valign="top">
<bold>0.9381</bold>
</td>
<td align="center" valign="top">
<bold>0.0077</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold value indicates the best value.</p>
</table-wrap-foot>
</table-wrap>
<p>This outstanding performance is attributed to the integration of Convolutional Neural Networks (CNNs) with a transformer module, which captures both local and global semantic features. The transformer component is crucial for calculating global semantic relationships, while the CNN module extracts local contextual features, resulting in a more robust feature representation. Additionally, the multi-query attention (MQA) module enriches feature diversity through supervised learning, enhancing overall model performance.</p>
<p>An analysis of false positives (FP) and false negatives (FN) further emphasizes MoNetViT&#x2019;s robustness. With high sensitivity (0.9424) and specificity (0.9424), MoNetViT significantly reduces both FP and FN compared to other models. For instance, while DeepLabV3+ achieves a Dice score of 0.6351, its sensitivity (0.6392) and specificity (0.9655) indicate a higher FN rate relative to MoNetViT. Similarly, Mini-Seg&#x2019;s balanced sensitivity (0.6268) and specificity (0.9537) suggest that it is more prone to FP and FN, impacting its reliability. In contrast, MoNetViT&#x2019;s ability to minimize FP and FN contributes to its high Dice score and overall segmentation accuracy.</p>
<p>Despite Inf-Net having the smallest model size (0.073&#x202F;MB) and TransFuse having the smallest parameter count (0.038&#x202F;M), MoNetViT outperforms these models in critical metrics. Its effectiveness stems from model design rather than sheer training data volume, highlighting MoNetViT&#x2019;s robustness and capacity to generalize effectively on the dataset. Paired <italic>T</italic>-tests indicated that MoNetViT&#x2019;s improvements were statistically significant (<italic>p</italic> &#x003C;&#x202F;0.05) compared to all models except U-Net++, Mini-Seg, and DeepLabV3+, suggesting that while MoNetViT is generally superior, these models are still competitive in certain tasks. The addition of FP and FN analysis strengthens these findings, demonstrating that MoNetViT achieves its exceptional performance by addressing key limitations in segmentation errors observed in other models.</p>
</sec>
<sec id="sec16">
<label>3.2</label>
<title>Component impact analysis</title>
<p>Several experiments were conducted to validate the functionality of the Multi Query Attention (MQA) and Fusion Feature Module (FFM), two crucial elements of the MoNetViT. <xref ref-type="fig" rid="fig1">Figure 1A</xref> depicts an architecture consisting of three stages. The segmentation performance of the MoNetViT model is considerably improved by the MQA module component, as the findings displayed in <xref ref-type="table" rid="tab4">Table 4</xref> reveal.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>MoNetViT ablation study.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="middle" char="&#x00D7;">Methods</th>
<th align="char" valign="middle" char="&#x00D7;">Loss</th>
<th align="char" valign="middle" char="&#x00D7;">Dice</th>
<th align="char" valign="middle" char="&#x00D7;">Sen.</th>
<th align="char" valign="middle" char="&#x00D7;">Spec.</th>
<th align="char" valign="middle" char="&#x00D7;">
<inline-formula>
<mml:math id="M250">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>&#x03B1;</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>
</th>
<th align="char" valign="middle" char="&#x00D7;">
<inline-formula>
<mml:math id="M251">
<mml:msubsup>
<mml:mi>E</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi mathvariant="italic">mean</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula>
</th>
<th align="char" valign="middle" char="&#x00D7;">MAE</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Backbone</td>
<td align="center" valign="top">0.0183</td>
<td align="center" valign="top">0.9584</td>
<td align="center" valign="top">0.9424</td>
<td align="center" valign="top">0.9424</td>
<td align="center" valign="top">0.9923</td>
<td align="center" valign="top">0.9381</td>
<td align="center" valign="top">0.0077</td>
</tr>
<tr>
<td align="left" valign="top">Backbone+MQA</td>
<td align="center" valign="top">0.0192</td>
<td align="center" valign="top">0.9558</td>
<td align="center" valign="top">0.9360</td>
<td align="center" valign="top">0.9360</td>
<td align="center" valign="top">0.9919</td>
<td align="center" valign="top">0.9342</td>
<td align="center" valign="top">0.0081</td>
</tr>
<tr>
<td align="left" valign="top">Backbone+FFM</td>
<td align="center" valign="top">0.0189</td>
<td align="center" valign="top">0.9566</td>
<td align="center" valign="top">0.9322</td>
<td align="center" valign="top">0.9322</td>
<td align="center" valign="top">0.9921</td>
<td align="center" valign="top">0.9354</td>
<td align="center" valign="top">0.0079</td>
</tr>
<tr>
<td align="left" valign="top">Backbone+MQA&#x202F;+&#x202F;FFM</td>
<td align="center" valign="top">
<bold>0.0115</bold>
</td>
<td align="center" valign="top">
<bold>0.9731</bold>
</td>
<td align="center" valign="top">
<bold>0.9686</bold>
</td>
<td align="center" valign="top">
<bold>0.9686</bold>
</td>
<td align="center" valign="top">
<bold>0.9951</bold>
</td>
<td align="center" valign="top">
<bold>0.9599</bold>
</td>
<td align="center" valign="top">
<bold>0.0049</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold value indicates the best value.</p>
</table-wrap-foot>
</table-wrap>
<p>Employing MQA and FFM in conjunction with the baseline improves segmentation performance. Specifically, integrating both MQA and FFM with the baseline resulted in improvements of 1.5 and 1.7% in the Dice coefficient, respectively. The findings indicate that using MQA and FFM enhances the encoder&#x2019;s and decoder&#x2019;s clarity, thereby further improving segmentation performance. The analysis shows that combining Backbone, FFM, and MQA results in the best performance across all metrics. This model has the lowest loss (0.0115) and the highest Dice coefficient (0.9731), indicating more accurate segmentation and fewer errors. It also achieves the best sensitivity (0.9686) and specificity (0.9686), accurately identifying both positive and negative cases. Additionally, it has the highest structural alignment (<inline-formula>
<mml:math id="M252">
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>&#x03B1;</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> = 0.9951) and precision (<inline-formula>
<mml:math id="M253">
<mml:msubsup>
<mml:mi>E</mml:mi>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi mathvariant="italic">mean</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> = 0.9599), along with the lowest Mean Absolute Error (0.0049). In comparison, other methods like Backbone alone, Backbone+FFM, and Backbone+MQA perform worse in various metrics, highlighting the benefits of using both FFM and MQA together.</p>
</sec>
<sec id="sec17">
<label>3.3</label>
<title>Parameter comparison</title>
<p>
<xref ref-type="fig" rid="fig4">Figure 4</xref> provides a detailed overview of the model performance. MoNetViT achieves superior accuracy while maintaining a relatively small parameter count. Specifically, it uses about half the parameters of U-Net (1.014&#x202F;M vs. 1.953&#x202F;M) and significantly fewer parameters than U-Net++ (7.783&#x202F;M) and DeepLabV3+ (13.324&#x202F;M). While Inf-Net (0.076&#x202F;M) and Mini-Seg (0.038&#x202F;M) have smaller parameter counts, and TransFuse uses the fewest (0.019&#x202F;M), MoNetViT achieves a much higher Dice score (0.9584) compared to these models. This demonstrates that MoNetViT not only optimizes model size and complexity but also outperforms other models, including transformer-based models like TransFuse, in terms of accuracy (<xref ref-type="fig" rid="fig5">Figure 5</xref>).</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Dice vs. number of parameters between different segmentation algorithms.</p>
</caption>
<graphic xlink:href="fcomp-07-1510252-g004.tif"/>
</fig>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>FFM and MQA improvement assessment.</p>
</caption>
<graphic xlink:href="fcomp-07-1510252-g005.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="sec18">
<label>4</label>
<title>Discussion</title>
<sec id="sec19">
<label>4.1</label>
<title>Comparison of multi-scale features</title>
<p>Numerous models for combining multi-scale features use conventional networks for object identification and semantic segmentation in image processing. For example, architectures like Feature Pyramid Network (FPN) and U-Net rely on three primary network paths: bottom-up, top-down, and horizontal connections. These paths allow the integration of high-level semantic information with low-level geometric data. In FPN, the bottom-up path extracts high-level features, while the top-down path applies upsampling to enhance semantic details at higher resolutions. Horizontal connections fuse low-level convolution features with high-level features, resulting in a more detailed representation of semantic information.</p>
<p>Nevertheless, FPN faces challenges due to its complex hierarchical structure. The computation of intermediary layers relies heavily on the higher-level layers, requiring the analysis of preceding layers to be completed before passing information to subsequent layers. This dependency can lead to inefficiencies in computation and integration. The proposed MQA approach addresses these limitations by enabling the simultaneous integration of lower-level map attributes, higher-level attribute maps, and edge attribute maps, streamlining the process and enhancing feature representation.</p>
<p>The current work offers a direct and efficient MQA approach and introduces a cascading multi input computational framework. The system employs a mechanism for attention to iteratively compute and use feature maps of varying sizes, directing the ultimate semantic segmentation process. The proposed MQA has the ability to combine multiple input and multiple scale features, and can be trained end-to-end with ground truth supervision. The proposed technique effectively leverages both low and high-resolution features and integrates a method of attention to successfully accomplish the segmentation job on ArUco markers.</p>
</sec>
<sec id="sec20">
<label>4.2</label>
<title>Comparison of different combination MQA and FFM</title>
<p>To evaluate the impact of MQA and FFM on MoNetViT&#x2019;s performance, a series of tests were conducted. The experiments utilized the same network backbone and implementation details to ensure consistency with previous studies. The results, as shown in the radar chart and table, compare the baseline &#x201C;Backbone,&#x201D; &#x201C;Backbone+MQA,&#x201D; &#x201C;Backbone+FFM,&#x201D; and &#x201C;Backbone+MQA&#x202F;+&#x202F;FFM&#x201D; configurations. The radar chart illustrates that the region representing &#x201C;Backbone+MQA&#x202F;+&#x202F;FFM&#x201D; (in red) is larger than those of other configurations, indicating superior performance across key metrics. Similarly, the table reinforces these findings, showing that &#x201C;Backbone+MQA&#x202F;+&#x202F;FFM&#x201D; achieves the highest Dice score (0.9731), sensitivity (0.9686), and specificity (0.9686), along with the lowest MAE (0.0049). These results suggest that integrating both MQA and FFM significantly enhances the backbone&#x2019;s performance, as the areas with MQA and FFM have a notably larger magnitude than those without these modules.</p>
<p>The contributions of the Multi-Query Attention (MQA) and Feature Fusion Module (FFM) to the segmentation performance were further validated through an ablation study. <xref ref-type="table" rid="tab4">Table 4</xref> highlights the significant improvements achieved by integrating these modules into the baseline model. Specifically, the Dice coefficient increased from 0.9584 for the baseline to 0.9558 (+1.7%) with MQA alone and 0.9566 (+1.9%) with FFM alone. When both modules were combined, the Dice coefficient reached 0.9731 (+3.4%), demonstrating their synergistic effect. Furthermore, sensitivity and specificity improved from 0.9424 each in the baseline to 0.9686 with the combined MQA and FFM setup, while the Mean Absolute Error (MAE) decreased from 0.0077 to 0.0049. These findings emphasize the critical role of MQA in enhancing multi-scale feature integration and FFM in refining feature clarity, resulting in superior segmentation performance. This robust improvement across key metrics underscores the effectiveness of the proposed MoNetViT architecture in addressing complex segmentation tasks.</p>
<p>MoNetViT&#x2019;s architecture demonstrates significant advantages through its dual-path encoder, which effectively balances local feature extraction using CNNs and global feature extraction via Transformers. This design allows the model to capture both fine-grained details and long-range dependencies, improving segmentation performance. Additionally, the integration of the Multi-Query Attention (MQA) module enhances multi-scale feature integration, enabling the model to better aggregate features across varying spatial scales. This contributes to improved segmentation accuracy, particularly in complex scenarios. Furthermore, MoNetViT&#x2019;s lightweight design minimizes computational demands, making it highly efficient for real-time applications without sacrificing performance. Compared to models like DeepLabV3+, which rely on more resource-intensive architectures, MoNetViT achieves a superior balance of accuracy and efficiency, reinforcing its suitability for deployment in resource-constrained environments.</p>
<p>While MoNetViT demonstrates strong performance, scalability to larger datasets may require optimization strategies, and its adaptability to diverse marker types needs further evaluation across different styles and conditions. Future experiments will focus on enhancing scalability and generalizability through transfer learning and broader dataset evaluations.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec21">
<label>5</label>
<title>Conclusion</title>
<p>This study introduces a novel model called MoNetViT, which utilizes fused CNNs and transformers to create a segmentation model for ArUco marker-infested regions. The picture features are extracted simultaneously utilizing Convolutional Neural Networks (CNNs) and transformers, resulting in a reduction in computing burden and model complexity, while enhancing the segmentation performance. Furthermore, this work introduces the multi-query attention (MQA) module as a means to enhance performance. The empirical findings demonstrate that MoNetViT outperforms the other approaches on the ArUco dataset. Further studies will concentrate on the influence of the merging of each model and on discovering strategies for decreasing the level of detail within the model. Future research will focus on enhancing the capabilities of MoNetViT to achieve even more robust outcomes. This includes exploring additional fusion methods to further optimize feature integration and segmentation accuracy. Another key direction is adapting the model for outdoor environments by incorporating GPS data, thereby extending its applicability to diverse navigation scenarios. Leveraging transfer learning techniques will also be prioritized to reduce training times and improve scalability, enabling the model to handle larger and more diverse datasets effectively. These advancements aim to broaden the utility and efficiency of MoNetViT, ensuring its suitability for a wide range of real-world applications.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec22">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found in the article/supplementary material.</p>
</sec>
<sec sec-type="author-contributions" id="sec23">
<title>Author contributions</title>
<p>LT: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. RG: Conceptualization, Formal analysis, Funding acquisition, Investigation, Supervision, Validation, Writing &#x2013; review &#x0026; editing. Prayitno: Conceptualization, Formal analysis, Funding acquisition, Investigation, Supervision, Validation, Writing &#x2013; review &#x0026; editing, Data curation.</p>
</sec>
<sec sec-type="funding-information" id="sec24">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<ack>
<p>The authors thank the Ministry of Education, Culture, Research, and Technology for the research grant and support provided for this doctoral dissertation. Additionally, we extend our thanks to Diponegoro University, especially the Doctoral Program of Information Systems, for their continuous support and valuable resources that have significantly contributed to the success of this research.</p>
</ack>
<sec sec-type="COI-statement" id="sec25">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec26">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="sec27">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abozeid</surname> <given-names>A.</given-names></name> <name><surname>Taloba</surname> <given-names>A.</given-names></name> <name><surname>Faiz Alwaghid</surname> <given-names>A.</given-names></name> <name><surname>Salem</surname> <given-names>M.</given-names></name> <name><surname>Elhadad</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>An efficient indoor localization based on deep attention learning model</article-title>. <source>Comput. Syst. Sci. Eng.</source> <volume>46</volume>, <fpage>2637</fpage>&#x2013;<lpage>2650</lpage>. doi: <pub-id pub-id-type="doi">10.32604/csse.2023.037761</pub-id>, PMID: <pub-id pub-id-type="pmid">39055887</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asadi Shamsabadi</surname> <given-names>E.</given-names></name> <name><surname>Xu</surname> <given-names>C.</given-names></name> <name><surname>Rao</surname> <given-names>A. S.</given-names></name> <name><surname>Nguyen</surname> <given-names>T.</given-names></name> <name><surname>Ngo</surname> <given-names>T.</given-names></name> <name><surname>Dias-da-Costa</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>Vision transformer-based autonomous crack detection on asphalt and concrete surfaces</article-title>. <source>Autom. Constr.</source> <volume>140</volume>:<fpage>104316</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.autcon.2022.104316</pub-id>, PMID: <pub-id pub-id-type="pmid">39834397</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bai</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Lian</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>D.</given-names></name></person-group> (<year>2019</year>). <article-title>Wearable travel aid for environment perception and navigation of visually impaired people</article-title>. <source>Electronics</source> <volume>8</volume>:<fpage>697</fpage>. doi: <pub-id pub-id-type="doi">10.3390/electronics8060697</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benmouna</surname> <given-names>B.</given-names></name> <name><surname>Pourdarbani</surname> <given-names>R.</given-names></name> <name><surname>Sabzi</surname> <given-names>S.</given-names></name> <name><surname>Fernandez-Beltran</surname> <given-names>R.</given-names></name> <name><surname>Garc&#x00ED;a-Mateos</surname> <given-names>G.</given-names></name> <name><surname>Molina-Mart&#x00ED;nez</surname> <given-names>J. M.</given-names></name></person-group> (<year>2023</year>). <article-title>Attention mechanisms in convolutional neural networks for nitrogen treatment detection in tomato leaves using hyperspectral images</article-title>. <source>Electronics</source> <volume>12</volume>:<fpage>22706</fpage>. doi: <pub-id pub-id-type="doi">10.3390/electronics12122706</pub-id>, PMID: <pub-id pub-id-type="pmid">39800344</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Lu</surname> <given-names>Y.</given-names></name> <name><surname>Yu</surname> <given-names>Q.</given-names></name> <name><surname>Luo</surname> <given-names>X.</given-names></name> <name><surname>Adeli</surname> <given-names>E.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>TransUNet: transformers make strong encoders for medical image segmentation</article-title>, <source>arXiv</source>. <fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2102.04306</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Tang</surname> <given-names>H.</given-names></name> <name><surname>Zhao</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Tan</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>CoTrFuse: a novel framework by fusing CNN and transformer for medical image segmentation</article-title>. <source>Phys. Med.</source> <volume>68</volume>:<fpage>175027</fpage>. doi: <pub-id pub-id-type="doi">10.1088/1361-6560/acede8</pub-id>, PMID: <pub-id pub-id-type="pmid">37605997</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>d&#x2019;Ascoli</surname> <given-names>S.</given-names></name> <name><surname>Touvron</surname> <given-names>H.</given-names></name> <name><surname>Leavitt</surname> <given-names>M. L.</given-names></name> <name><surname>Morcos</surname> <given-names>A. S.</given-names></name> <name><surname>Biroli</surname> <given-names>G.</given-names></name> <name><surname>Sagun</surname> <given-names>L.</given-names></name></person-group> (<year>2022</year>). <article-title>ConViT: improving vision transformers with soft convolutional inductive biases</article-title>. <source>J. Stat. Mech. Theor. Exp.</source> <volume>2022</volume>, <fpage>2286</fpage>&#x2013;<lpage>2296</lpage>. doi: <pub-id pub-id-type="doi">10.1088/1742-5468/ac9830</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Doppalapudi</surname> <given-names>S. A. I. K.</given-names></name></person-group> (<year>2023</year>). <source>Semantic image segmentation using transformers</source>.</citation></ref>
<ref id="ref9"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname> <given-names>A</given-names></name></person-group>. (<year>2021</year>). &#x201C;An image is WORTH 16X16 WORDS: transformers for image recognition at scale,&#x201D; in <italic>ICLR 2021 - 9th International Conference on Learning Representations</italic>.</citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname> <given-names>A.</given-names></name> <name><surname>Beyer</surname> <given-names>L.</given-names></name> <name><surname>Kolesnikov</surname> <given-names>A.</given-names></name> <name><surname>Weissenborn</surname> <given-names>D.</given-names></name> <name><surname>Zhai</surname> <given-names>X</given-names></name> <name><surname>Unterthiner</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2021</year>). &#x201C;<article-title>An image is Worth 16x16 Words: transformers for image recognition at scale</article-title>,&#x201D; <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>El-taher</surname> <given-names>F.</given-names></name> <name><surname>Taha</surname> <given-names>A.</given-names></name> <name><surname>Courtney</surname> <given-names>J.</given-names></name> <name><surname>Mckeever</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>A systematic review of urban navigation systems for visually impaired people</article-title>. <source>Sensors</source> <volume>21</volume>:<fpage>3103</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s21093103</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>D. P.</given-names></name> <name><surname>Tao</surname> <given-names>Z.</given-names></name> <name><surname>Ge-Peng</surname> <given-names>J.</given-names></name> <name><surname>Yi</surname> <given-names>Z.</given-names></name> <name><surname>Geng</surname> <given-names>C.</given-names></name> <name><surname>Huazhu</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Enhanced-alignment measure for binary foreground map evaluation</article-title>. <source>IJCAI.</source> <volume>12</volume>, <fpage>698</fpage>&#x2013;<lpage>704</lpage>. doi: <pub-id pub-id-type="doi">10.24963/ijcai.2018/97</pub-id></citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>D.-P.</given-names></name> <name><surname>Zhou</surname> <given-names>T.</given-names></name> <name><surname>Ji</surname> <given-names>G. P.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name> <name><surname>Fu</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Inf-net: automatic COVID-19 lung infection segmentation from CT images</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>39</volume>, <fpage>2626</fpage>&#x2013;<lpage>2637</lpage>. doi: <pub-id pub-id-type="doi">10.1109/tmi.2020.2996645</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fernando</surname> <given-names>N.</given-names></name> <name><surname>McMeekin</surname> <given-names>D. A.</given-names></name> <name><surname>Murray</surname> <given-names>I.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x2018;Route planning methods in indoor navigation tools for vision impaired persons: a systematic review&#x2019;, disability and rehabilitation</article-title>. <source>Assist. Technol.</source> <volume>18</volume>, <fpage>763</fpage>&#x2013;<lpage>782</lpage>. doi: <pub-id pub-id-type="doi">10.1080/17483107.2021.1922522</pub-id>, PMID: <pub-id pub-id-type="pmid">34043928</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>Q.</given-names></name> <name><surname>Lei</surname> <given-names>Y.</given-names></name> <name><surname>Xing</surname> <given-names>W.</given-names></name> <name><surname>He</surname> <given-names>C.</given-names></name> <name><surname>Wei</surname> <given-names>G.</given-names></name> <name><surname>Miao</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Evaluation of pulmonary edema using ultrasound imaging in patients with COVID-19 pneumonia based on a Non-Local Channel attention ResNet</article-title>. <source>Ultrasound in Medicine \&#x0026; Biology [Preprint].</source> <volume>48</volume>, <fpage>945</fpage>&#x2013;<lpage>953</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ultrasmedbio.2022.01.023</pub-id>, PMID: <pub-id pub-id-type="pmid">35277285</pub-id></citation></ref>
<ref id="ref15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>X.</given-names></name> <name><surname>Yang</surname> <given-names>K.</given-names></name> <name><surname>Fei</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>K.</given-names></name></person-group> (<year>2019</year>). <article-title>ACNET: attention based network to exploit complementary features for RGBD semantic segmentation</article-title>. <source>IEEE</source> <volume>21</volume>, <fpage>1440</fpage>&#x2013;<lpage>1444</lpage>. doi: <pub-id pub-id-type="doi">10.1109/icip.2019.8803025</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jeamwatthanachai</surname> <given-names>W.</given-names></name> <name><surname>Wald</surname> <given-names>M.</given-names></name> <name><surname>Wills</surname> <given-names>G.</given-names></name></person-group> (<year>2019</year>). <article-title>Indoor navigation by blind people: behaviors and challenges in unfamiliar spaces and buildings</article-title>. <source>Br. J. Vis. Impair.</source> <volume>37</volume>, <fpage>140</fpage>&#x2013;<lpage>153</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0264619619833723</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Karimi</surname> <given-names>D.</given-names></name> <name><surname>Dou</surname> <given-names>H.</given-names></name> <name><surname>Warfield</surname> <given-names>S. K.</given-names></name> <name><surname>Gholipour</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>Deep learning with Noisy labels: exploring techniques and remedies in medical image analysis</article-title>. <source>Med. Image Anal.</source> <volume>65</volume>:<fpage>101759</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.media.2020.101759</pub-id>, PMID: <pub-id pub-id-type="pmid">32623277</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>H. W.</given-names></name> <name><surname>Lee</surname> <given-names>S.</given-names></name> <name><surname>Yang</surname> <given-names>J. H.</given-names></name> <name><surname>Moon</surname> <given-names>Y.</given-names></name> <name><surname>Lee</surname> <given-names>J.</given-names></name> <name><surname>Moon</surname> <given-names>W. J.</given-names></name></person-group> (<year>2023</year>). <article-title>Cortical Iron accumulation as an imaging marker for neurodegeneration in clinical cognitive impairment Spectrum: a quantitative susceptibility mapping study</article-title>. <source>Korean J. Radiol.</source> <volume>24</volume>, <fpage>1131</fpage>&#x2013;<lpage>1141</lpage>. doi: <pub-id pub-id-type="doi">10.3348/kjr.2023.0490</pub-id>, PMID: <pub-id pub-id-type="pmid">37899522</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kubota</surname> <given-names>M.</given-names></name></person-group> (<year>2024</year>). <article-title>Snap: smartphone-based indoor navigation system for blind people via floor map analysis and intersection detection</article-title>. <source>Proc. ACM Hum. Comput. Int.</source> <volume>8</volume>, <fpage>1</fpage>&#x2013;<lpage>22</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3676522</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuriakose</surname> <given-names>B.</given-names></name> <name><surname>Shrestha</surname> <given-names>R.</given-names></name> <name><surname>Sandnes</surname> <given-names>F. E.</given-names></name></person-group> (<year>2021</year>). <article-title>Towards independent navigation with visual impairment: a prototype of a deep learning and smartphone-based assistant</article-title>. <source>Association for Computing Machinery.</source> <fpage>113</fpage>&#x2013;<lpage>114</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3453892.3464946</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kwak</surname> <given-names>J.</given-names></name> <name><surname>Sung</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>DeepLabV3-refiner-based semantic segmentation model for dense 3D point clouds</article-title>. <source>Remote Sens.</source> <volume>13</volume>:<fpage>165</fpage>. doi: <pub-id pub-id-type="doi">10.3390/rs13081565</pub-id>, PMID: <pub-id pub-id-type="pmid">39800344</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Lazebnik</surname> <given-names>S.</given-names></name> <name><surname>Schmid</surname> <given-names>C.</given-names></name> <name><surname>Ponce</surname> <given-names>J.</given-names></name></person-group> (<year>2006</year>). &#x201C;Beyond bags of features: spatial pyramid matching for recognizing natural scene categories,&#x201D; in <italic>2006 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR&#x2019;06)</italic>, 2169&#x2013;2178.</citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>J.</given-names></name> <name><surname>Yoon</surname> <given-names>W.</given-names></name> <name><surname>Kim</surname> <given-names>S.</given-names></name> <name><surname>Kim</surname> <given-names>D.</given-names></name> <name><surname>Kim</surname> <given-names>S.</given-names></name> <name><surname>So</surname> <given-names>C. H.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>BioBERT: a pre-trained biomedical language representation model for biomedical text mining</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>1234</fpage>&#x2013;<lpage>1240</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id>, PMID: <pub-id pub-id-type="pmid">31501885</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Lei</surname> <given-names>W.</given-names></name> <name><surname>Xia</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep learning based mineral image classification combined with visual attention mechanism</article-title>. <source>IEEE Access</source> <volume>9</volume>, <fpage>98091</fpage>&#x2013;<lpage>98109</lpage>. doi: <pub-id pub-id-type="doi">10.1109/access.2021.3095368</pub-id>, PMID: <pub-id pub-id-type="pmid">39573497</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mart&#x00ED;nez-Cruz</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>An outdoor navigation assistance system for visually impaired people in public transportation</article-title>. <source>IEEE Access</source> <volume>9</volume>, <fpage>130767</fpage>&#x2013;<lpage>130777</lpage>. doi: <pub-id pub-id-type="doi">10.1109/access.2021.3111544</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mehta</surname> <given-names>S.</given-names></name> <name><surname>Apple</surname> <given-names>M. R.</given-names></name></person-group> (<year>2022</year>). <article-title>MobileViT: light-weight, general-purpose, and Mobile-friendly vision transformer</article-title>. <source>ICLR</source> <volume>22</volume>:<fpage>3</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2110.02178</pub-id></citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mei</surname> <given-names>Y.</given-names></name> <name><surname>Fan</surname> <given-names>Y.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name> <name><surname>Huang</surname> <given-names>T. S.</given-names></name> <name><surname>Shi</surname> <given-names>H</given-names></name></person-group>. (<year>2020</year>). <article-title>Image super-resolution with cross-scale non-local attention and exhaustive self-exemplars mining</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arxiv.2006.01424</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Messaoudi</surname> <given-names>M. D.</given-names></name> <name><surname>Menelas</surname> <given-names>B. A. J.</given-names></name> <name><surname>Mcheick</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>Autonomous smart white cane navigation system for indoor usage</article-title>. <source>Technologies</source> <volume>8</volume>:<fpage>37</fpage>. doi: <pub-id pub-id-type="doi">10.3390/technologies8030037</pub-id>, PMID: <pub-id pub-id-type="pmid">39800344</pub-id></citation></ref>
<ref id="ref30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Misawa</surname> <given-names>N.</given-names></name> <name><surname>Yamaguchi</surname> <given-names>R.</given-names></name> <name><surname>Yamada</surname> <given-names>A.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Matsui</surname> <given-names>C.</given-names></name> <name><surname>Takeuchi</surname> <given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>Design methodology of compact edge vision transformer CiM considering non-volatile memory bit precision and memory error tolerance</article-title>. <source>Jap. J. Appl. Phys.</source> <volume>63</volume>:<fpage>03SP05</fpage>. doi: <pub-id pub-id-type="doi">10.35848/1347-4065/ad1bbd</pub-id>, PMID: <pub-id pub-id-type="pmid">38911013</pub-id></citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mujtaba</surname> <given-names>G.</given-names></name> <name><surname>Malik</surname> <given-names>A.</given-names></name> <name><surname>Ryu</surname> <given-names>E.</given-names></name></person-group> (<year>2022</year>). <article-title>LTC-SUM: lightweight client-driven personalized video summarization framework using 2D CNN</article-title>. <source>IEEE access</source> <volume>10</volume>, <fpage>103041</fpage>&#x2013;<lpage>103055</lpage>. doi: <pub-id pub-id-type="doi">10.1109/access.2022.3209275</pub-id>, PMID: <pub-id pub-id-type="pmid">39573497</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Petit</surname> <given-names>O.</given-names></name> <name><surname>Thome</surname> <given-names>N.</given-names></name> <name><surname>Rambour</surname> <given-names>C.</given-names></name> <name><surname>Themyr</surname> <given-names>L.</given-names></name> <name><surname>Collins</surname> <given-names>T.</given-names></name> <name><surname>Soler</surname> <given-names>L.</given-names></name></person-group> (<year>2021</year>). &#x201C;<article-title>U-net transformer: self and cross attention for medical image segmentation</article-title>&#x201D; in <source>Machine learning in medical imaging</source>. eds. <person-group person-group-type="editor"><name><surname>Lian</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>267</fpage>&#x2013;<lpage>276</lpage>.</citation></ref>
<ref id="ref33"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Qi</surname> <given-names>M.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name></person-group> (<year>2023</year>). <article-title>Dimensional emotion recognition based on two stream CNN fusion attention mechanism</article-title>. <source>Third International Conference on Sensors and Information Technology (ICSI 2023)</source>. (Eds.). <person-group person-group-type="editor"><name><surname>Kannan</surname> <given-names>H.</given-names></name> <name><surname>Hemanth</surname> <given-names>J.</given-names></name></person-group>. <publisher-name>International Society for Optics and Photonics</publisher-name>.</citation></ref>
<ref id="ref34"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Ronneberger</surname> <given-names>O.</given-names></name> <name><surname>Fischer</surname> <given-names>P.</given-names></name> <name><surname>Brox</surname> <given-names>T.</given-names></name></person-group> (<year>2015</year>). <source>U-net: convolutional networks for biomedical image segmentation</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation></ref>
<ref id="ref35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sandler</surname> <given-names>M.</given-names></name> <name><surname>Howard</surname> <given-names>A.</given-names></name> <name><surname>Zhu</surname> <given-names>M.</given-names></name> <name><surname>Zhmoginov</surname> <given-names>A.</given-names></name> <name><surname>Chen</surname> <given-names>L.-C.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>MobileNetV2: inverted residuals and linear bottlenecks</article-title>,&#x201D; in <source>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, <fpage>4510</fpage>&#x2013;<lpage>4520</lpage>.</citation></ref>
<ref id="ref36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shah</surname> <given-names>P. A.</given-names></name> <name><surname>Hessah</surname> <given-names>A. R. A. A.</given-names></name> <name><surname>Rayan</surname> <given-names>H. M. A. A.</given-names></name></person-group> (<year>2023</year>). <source>Machine learning-based smart assistance system for the visually impaired</source>. <fpage>139</fpage>&#x2013;<lpage>144</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ITT59889.2023.10184257</pub-id></citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>S.</given-names></name> <name><surname>Yue</surname> <given-names>X.</given-names></name> <name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Torr</surname> <given-names>P. H. S.</given-names></name> <name><surname>Bai</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x2018;Patch-based separable transformer for visual recognition</article-title>. <source>IEEE transactions on pattern analysis and machine intelligence</source> <volume>22</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.1109/tpami.2022.3231725</pub-id>, PMID: <pub-id pub-id-type="pmid">37015401</pub-id></citation></ref>
<ref id="ref38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tao</surname> <given-names>Y.</given-names></name> <name><surname>Ganz</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>Simulation framework for evaluation of indoor navigation systems</article-title>. <source>IEEE access</source> <volume>8</volume>, <fpage>20028</fpage>&#x2013;<lpage>20042</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2020.2968435</pub-id>, PMID: <pub-id pub-id-type="pmid">39573497</pub-id></citation></ref>
<ref id="ref39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thakur</surname> <given-names>P. S.</given-names></name> <name><surname>Chaturvedi</surname> <given-names>S.</given-names></name> <name><surname>Khanna</surname> <given-names>P.</given-names></name> <name><surname>Sheorey</surname> <given-names>T.</given-names></name> <name><surname>Ojha</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Vision transformer meets convolutional neural network for plant disease classification</article-title>. <source>Eco. Inform.</source> <volume>77</volume>:<fpage>102245</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ecoinf.2023.102245</pub-id>, PMID: <pub-id pub-id-type="pmid">39834397</pub-id></citation></ref>
<ref id="ref40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Theodorou</surname> <given-names>P.</given-names></name> <name><surname>Tsiligkos</surname> <given-names>K.</given-names></name> <name><surname>Meliones</surname> <given-names>A.</given-names></name> <name><surname>Tsigris</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>An extended usability and UX evaluation of a Mobile application for the navigation of individuals with blindness and visual impairments indoors: an evaluation approach combined with training sessions</article-title>. <source>Br. J. Vis. Impair.</source> <volume>42</volume>, <fpage>86</fpage>&#x2013;<lpage>123</lpage>. doi: <pub-id pub-id-type="doi">10.1177/02646196221131739</pub-id>, PMID: <pub-id pub-id-type="pmid">39807426</pub-id></citation></ref>
<ref id="ref41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaswani</surname> <given-names>A.</given-names></name> <name><surname>Shazeer</surname> <given-names>N.</given-names></name> <name><surname>Parmar</surname> <given-names>N.</given-names></name> <name><surname>Uszkoreit</surname> <given-names>J.</given-names></name> <name><surname>Jones</surname> <given-names>L.</given-names></name> <name><surname>Gomez</surname> <given-names>A. N.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Attention is all youneed</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1706.03762</pub-id></citation></ref>
<ref id="ref43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Gupta</surname> <given-names>A.</given-names></name> <name><surname>He</surname> <given-names>K</given-names></name></person-group>. (<year>2018</year>). &#x201C;<article-title>Non-local neural networks</article-title>,&#x201D;in <source>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, <fpage>7794</fpage>&#x2013;<lpage>7803</lpage>.</citation></ref>
<ref id="ref42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Yi</surname> <given-names>B.</given-names></name> <name><surname>Yu</surname> <given-names>K.</given-names></name></person-group> (<year>2022</year>). <article-title>Exploration and research about key technologies concerning deep learning models targeting Mobile terminals</article-title>. <source>J. Phys. Conf. Series</source> <volume>2303</volume>:<fpage>012086</fpage>. doi: <pub-id pub-id-type="doi">10.1088/1742-6596/2303/1/012086</pub-id></citation></ref>
<ref id="ref44"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Jianbo</surname> <given-names>W.</given-names></name> <name><surname>Pengfei</surname> <given-names>M.</given-names></name> <name><surname>Fuyong</surname> <given-names>W.</given-names></name> <name><surname>Wenbao</surname> <given-names>D</given-names></name></person-group>. (<year>2019</year>). &#x201C;<article-title>Evaluation for parachute reliability based on fiducial inference and Bayesian network</article-title>,&#x201D; in <source>Proceedings of the 2018 International Conference on Mathematics, Modeling, Simulation and Statistics Application (MMSSA 2018)</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Atlantis Press</publisher-name>.</citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname> <given-names>F.</given-names></name> <name><surname>Wang</surname> <given-names>M.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>DFAM-DETR: deformable feature based attention mechanism DETR on slender object detection</article-title>. <source>IEICE Trans. Inf. Syst.</source> <volume>E106</volume>, <fpage>401</fpage>&#x2013;<lpage>409</lpage>. doi: <pub-id pub-id-type="doi">10.1587/transinf.2022edp7111</pub-id>, PMID: <pub-id pub-id-type="pmid">12739966</pub-id></citation></ref>
<ref id="ref46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xia</surname> <given-names>Z.</given-names></name> <name><surname>Kim</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>Enhancing mask transformer with auxiliary convolution layers for semantic segmentation</article-title>. <source>Sensors</source> <volume>23</volume>:<fpage>20581</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s23020581</pub-id>, PMID: <pub-id pub-id-type="pmid">36679377</pub-id></citation></ref>
<ref id="ref47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yeung</surname> <given-names>M.</given-names></name> <name><surname>Sala</surname> <given-names>E.</given-names></name> <name><surname>Sch&#x00F6;nlieb</surname> <given-names>C. B.</given-names></name> <name><surname>Rundo</surname> <given-names>L.</given-names></name></person-group> (<year>2022</year>). <article-title>Unified focal loss: generalising dice and cross entropy-based losses to handle class imbalanced medical image segmentation</article-title>. <source>Comput. Med. Imaging Graph.</source> <volume>95</volume>:<fpage>102026</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compmedimag.2021.102026</pub-id>, PMID: <pub-id pub-id-type="pmid">34953431</pub-id></citation></ref>
<ref id="ref49"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>Q.</given-names></name></person-group> (<year>2021</year>) &#x2018;<article-title>TransFuse: fusing transformers and CNNs for medical image segmentation</article-title>&#x2019;, in <person-group person-group-type="editor"><name><surname>Bruijne</surname> <given-names>M.</given-names><prefix>de</prefix></name></person-group> <source>Medical Image Computing and Computer Assisted Intervention -- MICCAI 2021</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>, <fpage>14</fpage>&#x2013;<lpage>24</lpage></citation></ref>
<ref id="ref48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Gao</surname> <given-names>Q.</given-names></name> <name><surname>Liu</surname> <given-names>L.</given-names></name> <name><surname>He</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>A high-quality Rice leaf disease image data augmentation method based on a dual GAN</article-title>. <source>IEEE access</source> <volume>11</volume>, <fpage>21176</fpage>&#x2013;<lpage>21191</lpage>. doi: <pub-id pub-id-type="doi">10.1109/access.2023.3251098</pub-id>, PMID: <pub-id pub-id-type="pmid">39573497</pub-id></citation></ref>
<ref id="ref50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Zhong</surname> <given-names>Y</given-names></name></person-group>. (<year>2024</year>). <source>Are transformers more suitable for plant disease identification than convolutional neural networks?</source>. doi: <pub-id pub-id-type="doi">10.21203/rs.3.rs-4284240/v1</pub-id></citation></ref>
<ref id="ref53"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>L.</given-names></name> <name><surname>Zhu</surname> <given-names>M.</given-names></name> <name><surname>Xiong</surname> <given-names>D.</given-names></name> <name><surname>Ouyang</surname> <given-names>L.</given-names></name> <name><surname>Ouyang</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>MR image reconstruction via non-local attention networks</article-title>. <source>Fourteenth International Conference on Graphics and Image Processing (ICGIP 2022)</source>. (Eds.) <person-group person-group-type="editor"><name><surname>Xiao</surname> <given-names>L.</given-names></name> <name><surname>Xue</surname> <given-names>J</given-names></name></person-group>. <publisher-name>International Society for Optics and Photonics</publisher-name>.</citation></ref>
<ref id="ref51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>L.</given-names></name> <name><surname>Kang</surname> <given-names>Z.</given-names></name> <name><surname>Zhou</surname> <given-names>M.</given-names></name> <name><surname>Yang</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Cao</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>CMANet: cross-modality attention network for indoor-scene semantic segmentation</article-title>. <source>Sensors</source> <volume>22</volume>:<fpage>8520</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s22218520</pub-id>, PMID: <pub-id pub-id-type="pmid">36366217</pub-id></citation></ref>
<ref id="ref52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>Z</given-names></name> <name><surname>Mengde</surname> <given-names>X.</given-names></name> <name><surname>Song</surname> <given-names>B.</given-names></name> <name><surname>Tengteng</surname> <given-names>H.</given-names></name> <name><surname>Xiang</surname> <given-names>B</given-names></name></person-group>. (<year>2019</year>). &#x201C;<article-title>Asymmetric non-local neural networks for semantic segmentation</article-title>,&#x201D; in <source>2019 IEEE/CVF International Conference on Computer Vision (ICCV)</source>, <fpage>593</fpage>&#x2013;<lpage>602</lpage>.</citation></ref>
</ref-list>
</back>
</article>