<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Comput. Neurosci.</journal-id>
<journal-title>Frontiers in Computational Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Comput. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-5188</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fncom.2025.1513059</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>MUNet: a novel framework for accurate brain tumor segmentation combining UNet and mamba networks</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Yang</surname> <given-names>Lijuan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Dong</surname> <given-names>Qiumei</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lin</surname> <given-names>Da</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2762528/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Tian</surname> <given-names>Chunfang</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>L&#x000FC;</surname> <given-names>Xinliang</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2894998/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Rheumatology, Inner Mongolia Autonomous Region Hospital of Traditional Chinese Medicine</institution>, <addr-line>Hohhot</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>College of Traditional Chinese Medicine, Inner Mongolia Medical University</institution>, <addr-line>Hohhot</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>School of Mathematical Sciences, Inner Mongolia University</institution>, <addr-line>Hohhot</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Oncology, Inner Mongolia Autonomous Region Hospital of Traditional Chinese Medicine</institution>, <addr-line>Hohhot</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Chenglong Zou, Peking University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Dulani Meedeniya, University of Moratuwa, Sri Lanka</p>
<p>Guang Chen, Peking University, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Xinliang L&#x000FC; <email>lxl230081&#x00040;sina.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>01</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>19</volume>
<elocation-id>1513059</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>01</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Yang, Dong, Lin, Tian and L&#x000FC;.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Yang, Dong, Lin, Tian and L&#x000FC;</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Brain tumors are one of the major health threats to humans, and their complex pathological features and anatomical structures make accurate segmentation and detection crucial. However, existing models based on Transformers and Convolutional Neural Networks (CNNs) still have limitations in medical image processing. While Transformers are proficient in capturing global features, they suffer from high computational complexity and require large amounts of data for training. On the other hand, CNNs perform well in extracting local features but have limited performance when handling global information. To address these issues, this paper proposes a novel network framework, MUNet, which combines the advantages of UNet and Mamba, specifically designed for brain tumor segmentation. MUNet introduces the SD-SSM module, which effectively captures both global and local features of the image through selective scanning and state-space modeling, significantly improving segmentation accuracy. Additionally, we design the SD-Conv structure, which reduces feature redundancy without increasing model parameters, further enhancing computational efficiency. Finally, we propose a new loss function that combines mIoU loss, Dice loss, and Boundary loss, which improves segmentation overlap, similarity, and boundary accuracy from multiple perspectives. Experimental results show that, on the BraTS2020 dataset, MUNet achieves DSC values of 0.835, 0.915, and 0.823 for enhancing tumor (ET), whole tumor (WT), and tumor core (TC), respectively, and Hausdorff95 scores of 2.421, 3.755, and 6.437. On the BraTS2018 dataset, MUNet achieves DSC values of 0.815, 0.901, and 0.815, with Hausdorff95 scores of 4.389, 6.243, and 6.152, all outperforming existing methods and achieving significant performance improvements. Furthermore, when validated on the independent LGG dataset, MUNet demonstrated excellent generalization ability, proving its effectiveness in various medical imaging scenarios. The code is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/Dalin1977331/MUNet">https://github.com/Dalin1977331/MUNet</ext-link>.</p></abstract>
<kwd-group>
<kwd>brain tumor segmentation</kwd>
<kwd>deep learning</kwd>
<kwd>MUNet</kwd>
<kwd>SD-SSM module</kwd>
<kwd>medical image analysis</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="6"/>
<equation-count count="24"/>
<ref-count count="43"/>
<page-count count="14"/>
<word-count count="8863"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Brain tumors are a type of malignant or benign tumor originating from brain cells or metastasizing from other parts of the body, posing a significant threat to human health (Michael et al., <xref ref-type="bibr" rid="B27">2021</xref>; Rezaei, <xref ref-type="bibr" rid="B32">2021</xref>). They exhibit diverse pathological manifestations and progress rapidly, often leading to severe neurological dysfunctions that affect the quality of life and even endanger the lives of patients (Abhisheka et al., <xref ref-type="bibr" rid="B1">2023</xref>; Ibrahim et al., <xref ref-type="bibr" rid="B19">2020</xref>). Due to the heterogeneity, deep location, and complex anatomical structure of brain tumors, early and accurate diagnosis and treatment are crucial for improving prognosis and therapeutic efficacy.</p>
<p>With the widespread application of deep learning in medical image analysis, researchers have leveraged its powerful feature extraction capabilities to achieve automatic segmentation and detection of brain tumors. Deep learning models can utilize large amounts of MRI data to learn the complex features of tumors, enabling accurate and efficient analysis and identification (Zebari et al., <xref ref-type="bibr" rid="B41">2020</xref>; Chaudhury et al., <xref ref-type="bibr" rid="B10">2022</xref>). Notably, structures like UNet have achieved remarkable results in segmentation tasks, making automated detection and segmentation of brain tumors possible. Compared to traditional manual feature extraction methods, these deep learning-based approaches can more precisely capture the shape and boundary characteristics of tumors, providing a robust tool for clinical assistance (Soulami et al., <xref ref-type="bibr" rid="B33">2021</xref>; Houssein et al., <xref ref-type="bibr" rid="B17">2021</xref>).</p>
<p>In recent years, deep learning models based on Transformers and Convolutional Neural Networks (CNNs) have made significant advancements in the field of image segmentation. Transformer architectures, with their self-attention mechanisms, excel at capturing global features in images (Wang et al., <xref ref-type="bibr" rid="B36">2023</xref>). Complementarily, CNNs have unique advantages in extracting local features, enabling the capture of spatial details within images. However, these methods also have certain limitations. CNN models face challenges in processing global information and may easily overlook long-range spatial correlations (Zhu et al., <xref ref-type="bibr" rid="B43">2024</xref>). On the other hand, while Transformers are proficient at capturing global features, their high computational complexity and demand for large-scale training data limit their application in medical image segmentation (Amgad et al., <xref ref-type="bibr" rid="B3">2022</xref>).</p>
<p>As an emerging deep learning structure, the Mamba network has achieved significant success in the field of computer vision. With its efficient feature extraction capabilities and modular design, Mamba has demonstrated strong performance advantages in tasks such as image segmentation, object detection, and image classification (Zhang et al., <xref ref-type="bibr" rid="B42">2024</xref>). Compared to traditional CNN models, the Mamba network excels in multi-scale feature fusion and contextual information capture, allowing it to better adapt to the complexity and diversity of visual data. Moreover, the Mamba structure significantly reduces computational complexity relative to Transformers (Badiezadeh et al., <xref ref-type="bibr" rid="B4">2024</xref>). However, medical images often contain complex textures and structures, especially MRI brain tumor images, which exhibit considerable heterogeneity and irregularity in their internal features. The variation in shape, size, location, and contrast of tumors relative to surrounding normal tissues makes feature extraction highly complex, posing challenges for the Mamba network in accurately segmenting and identifying tumor boundaries (Tang et al., <xref ref-type="bibr" rid="B34">2024</xref>).</p>
<p>In this paper, we propose a novel network framework named MUNet, which combines the advantages of Unet and Mamba, specifically designed for brain tumor segmentation. To achieve this, we design a new SD-SSM block structure that leverages selective scanning and state space modeling to capture both global and local features of the image. Moreover, without increasing the number of parameters, we introduce the SD-Conv structure, which consists of SCConv (Spatial and Channel Reconstruction Convolution) and Depthwise Separable Convolution, aiming to reduce feature redundancy and improve model efficiency. In MUNet, we apply skip connections to the SD-SSM block, fusing the features of the encoder and decoder to retain multi-scale information. Additionally, we design a new loss function that combines mIoU, Dice, and Boundary losses to optimize the overlap, similarity, and boundary accuracy of the segmentation.</p>
<p>Key contributions of this paper include:</p>
<list list-type="bullet">
<list-item><p>This paper proposes an innovative framework, MUNet, which combines Unet and Mamba, specifically for brain tumor segmentation. By fully integrating the advantages of both, MUNet achieves more precise and efficient segmentation of tumor regions.</p></list-item>
<list-item><p>This paper introduces a new SSM-based structure called the SD-SSM Block. Utilizing selective scanning and dual-channel feature extraction, it effectively captures multi-scale global and local features of the image, enhancing segmentation performance.</p></list-item>
<list-item><p>This paper presents the SD-Conv structure, which combines SCConv and DW Conv to compress redundant information between features without increasing the number of model parameters, thereby improving the efficiency of feature extraction.</p></list-item>
<list-item><p>For the task of brain tumor segmentation, this paper designs a novel loss function that combines mIoU loss, Dice loss, and Boundary loss. This approach optimizes the segmentation&#x00027;s overlap, similarity, and boundary accuracy from multiple perspectives, thereby enhancing the performance of MUNet in brain tumor segmentation.</p></list-item>
</list>
<p>The remaining structure of this paper is as follows: Section 2 presents the related work, introducing previous works on brain tumor segmentation using Unet and Mamba. Section 3 covers the methodology, providing a detailed explanation of the MUNet model concept. Section 4 describes the experiments, including comparative experiments and ablation studies. Finally, the conclusion summarizes the entire paper.</p>
</sec>
<sec id="s2">
<title>2 Related works</title>
<sec>
<title>2.1 U-Net network and its innovative evolution</title>
<p>In recent years, the U-Net network has made significant progress in the field of medical image analysis. As a fully convolutional neural network (FCN) (Ho et al., <xref ref-type="bibr" rid="B16">2021</xref>), U-Net employs its encoder-decoder symmetric structure to efficiently extract and fuse local and global features from images, showing exceptional performance in image segmentation tasks (Futrega et al., <xref ref-type="bibr" rid="B15">2021</xref>).</p>
<p>To further enhance the performance of U-Net in medical image segmentation, TransUNet (Chen et al., <xref ref-type="bibr" rid="B11">2021</xref>) combines U-Net with the Vision Transformer (ViT) to propose a hybrid segmentation model based on Transformers. TransUNet builds upon the U-Net encoder and utilizes the self-attention mechanism of Transformers to model global features in images. The ViT module (Li et al., <xref ref-type="bibr" rid="B23">2023b</xref>) introduces patch-level feature capturing during the encoding process, leveraging the global attention mechanism to capture long-range dependencies, thereby enhancing the model&#x00027;s understanding of global context in images. SwinUNet (Cao et al., <xref ref-type="bibr" rid="B9">2022</xref>) further integrates U-Net with the Swin Transformer, proposing a more efficient segmentation framework. Swin Transformer is a hierarchical vision transformer model that introduces a sliding window attention mechanism, effectively capturing long-range dependencies in images while reducing computational complexity. By embedding Swin Transformer modules into the encoder and decoder parts of U-Net, SwinUNet enhances the model&#x00027;s multi-scale feature extraction capabilities and global context modeling abilities. This model demonstrates excellent performance in medical image segmentation tasks, particularly for images with rich texture details and complex structures (Walsh et al., <xref ref-type="bibr" rid="B35">2022</xref>).</p>
<p>To further explore the potential of U-Net, the U-Mamba (Lee and Kim, <xref ref-type="bibr" rid="B21">2024</xref>) structure was developed. U-Mamba combines U-Net with the Mamba network, leveraging Mamba&#x00027;s strengths in feature extraction and multi-scale information fusion to improve segmentation accuracy. With its modular design, the Mamba (Xu et al., <xref ref-type="bibr" rid="B40">2024</xref>) network can be flexibly integrated into U-Net&#x00027;s encoder and decoder, enabling the comprehensive extraction and fusion of features at different scales. Moreover, the Mamba network exhibits strong generalization capabilities when handling complex textures and structures, allowing U-Mamba to achieve more precise segmentation results in medical imaging. Similarly, Mamba-Unet integrates U-Net with Vision Mamba (VMamba) (Zhu et al., <xref ref-type="bibr" rid="B43">2024</xref>), fully combining U-Net&#x00027;s context information capturing abilities with the feature expression advantages of VMamba, thus proposing an efficient network suitable for medical image segmentation (Patro and Agneeswaran, <xref ref-type="bibr" rid="B30">2024</xref>).</p>
<p>In this paper, we propose a novel framework that integrates the Mamba structure with UNet. Unlike existing approaches, we introduce the SD-SSM Block, which captures multi-scale features through selective scanning and dual-channel feature extraction. This design enables the model to better balance global feature modeling with detail preservation. Furthermore, we present the SD-Conv structure, which effectively reduces feature redundancy without increasing the number of parameters. This enhancement improves the efficiency of feature representation, enabling the model to achieve superior accuracy and performance in brain tumor segmentation.</p>
</sec>
<sec>
<title>2.2 Application of deep learning models in brain tumor segmentation</title>
<p>In recent years, brain tumor segmentation models have made significant progress in the field of deep learning (Magadza and Viriri, <xref ref-type="bibr" rid="B24">2021</xref>). ResUNet, for instance, is a model that combines the residual network (ResNet) (Maji et al., <xref ref-type="bibr" rid="B25">2022</xref>) with the U-Net structure. By introducing residual modules, it effectively addresses the gradient vanishing problem in deep networks, significantly enhancing model stability and convergence speed, thus improving the segmentation capability for complex brain tumor features. However, while this residual structure increases the model&#x00027;s expressive capacity, it also introduces higher computational costs, requiring more memory and longer training times. Models based on DenseNet (Belaid et al., <xref ref-type="bibr" rid="B7">2024</xref>) achieve efficient feature transmission through dense connections, fully leveraging multi-level features to improve segmentation accuracy, especially excelling in capturing tumor edge details. Nonetheless, the computational complexity and memory requirements brought by dense connections limit their application in resource-constrained environments. Attention Gated Networks (Chinnam et al., <xref ref-type="bibr" rid="B12">2022</xref>) introduce an attention mechanism, utilizing attention gates to focus on important features related to target regions, significantly improving focus on target areas and enhancing segmentation accuracy in complex backgrounds for brain tumors. However, the incorporation of the attention mechanism also increases the network&#x00027;s complexity, leading to longer inference times. SegNet (Almotairi et al., <xref ref-type="bibr" rid="B2">2020</xref>), as an encoder-decoder structured model, relies on max-pooling indices for upsampling, maintaining high computational efficiency and exhibiting good real-time performance in brain tumor segmentation, making it suitable for scenarios with high requirements for inference speed. However, compared to other models based on advanced convolutional structures, SegNet shows slightly lower accuracy in segmenting complex tumor morphologies. Residual Attention Networks combine residual connections and attention mechanisms to enhance the model&#x00027;s ability to express target region features while capturing both global and local information, and addressing gradient issues in deep network training (Ranjbarzadeh et al., <xref ref-type="bibr" rid="B31">2021</xref>). However, due to the introduction of residual and attention modules, this model has high computational resource requirements during inference and can exhibit certain inference delays (Jyothi and Singh, <xref ref-type="bibr" rid="B20">2023</xref>). In addition, an interpretable model based on U-Net and DenseNet has been designed for the segmentation and classification of brain tumors. This model enhances interpretability and transparency by generating heat maps that highlight the contribution of each region of the input image to the classification output (Wijethilake et al., <xref ref-type="bibr" rid="B38">2021</xref>). This approach not only improves the model&#x00027;s interpretability but also increases its trustworthiness in clinical diagnosis. Although these techniques provide new insights for tumor survival prediction, they also face several challenges. For example, existing models still suffer from poor interpretability and limited generalization ability, which restrict their widespread application in clinical practice. By combining imaging data with genomic information, more dimensions of data can be leveraged to provide a more reliable survival analysis of tumors, further enhancing the accuracy of predictions (Dasanayaka et al., <xref ref-type="bibr" rid="B14">2022b</xref>).</p>
<p>In this paper, we specifically designed MUNet for the task of brain tumor segmentation, overcoming the limitations of traditional CNN and Transformer architectures. MUNet effectively captures both local and global features of tumors, enhancing feature representation while also reducing computational complexity.</p>
</sec>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methods</title>
<sec>
<title>3.1 Preliminaries</title>
<p>State Space Models (SSM) (Zhu et al., <xref ref-type="bibr" rid="B43">2024</xref>) are a framework for modeling sequential data and are capable of capturing long-range dependencies. SSM is widely used in visual tasks for efficiently processing image sequences. It maps input sequences into a hidden state space and models sequences recursively. This section will introduce the SSM modeling process from three aspects: state updates, output generation, and efficient computation.</p>
<p><bold>State space representation</bold> The basic form of an SSM uses the state vector <italic>h</italic>(<italic>t</italic>) &#x02208; &#x0211D;<sup><italic>N</italic></sup> to represent the hidden state, mapping an input sequence <italic>x</italic>(<italic>t</italic>) &#x02208; &#x0211D; to an output sequence <italic>y</italic>(<italic>t</italic>) &#x02208; &#x0211D;. The state update equation and output equation are as follows:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>A</mml:mi><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>B</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>C</mml:mi><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where: <italic>A</italic> &#x02208; &#x0211D;<sup><italic>N</italic>&#x000D7;<italic>N</italic></sup> is the state transition matrix; <italic>B</italic> &#x02208; &#x0211D;<sup><italic>N</italic>&#x000D7;1</sup> is the input mapping matrix; <italic>C</italic> &#x02208; &#x0211D;<sup>1 &#x000D7; <italic>N</italic></sup> is the output mapping matrix.</p>
<p><bold>Discretization and time scale</bold> SSM is often a discretized version of a continuous system, introducing a time-scale parameter &#x00394; to convert a continuous-time state space into a discrete-time state space. To achieve this transformation, a Zero Order Hold (ZOH) is introduced:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mrow><mml:mover accent='true'><mml:mi>A</mml:mi><mml:mo>&#x000AF;</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mi>exp</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mo>&#x00394;</mml:mo><mml:mi>A</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mover accent="true"><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>&#x00394;</mml:mo><mml:mi>A</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>&#x00394;</mml:mo><mml:mi>A</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>I</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:mo>&#x00394;</mml:mo><mml:mi>B</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x00100; and <inline-formula><mml:math id="M5"><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow></mml:math></inline-formula> are the discretized state transition matrix and input mapping matrix, respectively. The state update equation in the discrete form is:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M6"><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mover accent='true'><mml:mi>A</mml:mi><mml:mo>&#x000AF;</mml:mo></mml:mover><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mover accent='true'><mml:mi>B</mml:mi><mml:mo>&#x000AF;</mml:mo></mml:mover><mml:msub><mml:mi>x</mml:mi><mml:mi>t</mml:mi></mml:msub><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<disp-formula id="E6"><label>(6)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>C</mml:mi><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><bold>Convolutional form and efficient computation</bold> In SSM, the state update process can be converted into a convolutional kernel form through convolution operations. Assuming the input sequence has a length of <italic>M</italic>, the convolution kernel <italic>K</italic> is represented as:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M8"><mml:mrow><mml:mi>K</mml:mi><mml:mo>=</mml:mo><mml:mo stretchy='false'>[</mml:mo><mml:mi>C</mml:mi><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:mover accent='true'><mml:mi>A</mml:mi><mml:mo>&#x000AF;</mml:mo></mml:mover><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:mi>C</mml:mi><mml:msup><mml:mover accent='true'><mml:mi>A</mml:mi><mml:mo>&#x000AF;</mml:mo></mml:mover><mml:mrow><mml:mi>M</mml:mi><mml:mo>&#x02212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mi>B</mml:mi><mml:mo stretchy='false'>]</mml:mo><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula>
<p>The output sequence of the SSM can then be computed using the convolution operation as follows:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mi>x</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x0002A; denotes the convolution operation.</p>
<p><bold>2D selective scan</bold> The traditional SSM are primarily designed for one-dimensional sequential data, which limits their ability to effectively capture the spatial information inherent in visual tasks. To overcome this challenge, a two-dimensional selective scanning (SS2D) method is introduced to model the 2D features in visual data effectively.</p>
<p>SS2D first divides the input image into a series of patches and arranges them in four directions: left to right, right to left, top to bottom, and bottom to top, generating four independent feature sequences. Let the original feature be <italic>z</italic> and the direction index be <italic>i</italic>. Each directional feature sequence can be represented as:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">expand</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>z</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>z</italic><sub><italic>i</italic></sub> is the feature sequence in the <italic>i</italic>-th direction, and the function expand represents the operation of arranging image patches according to the direction <italic>i</italic>.</p>
<p>In this way, SS2D achieves a global receptive field without significantly increasing the computational complexity, enabling the model to capture the global context of the image. Each generated feature sequence <italic>z</italic><sub><italic>i</italic></sub> is then processed through the selective scanning state space model, which performs feature extraction and modeling to obtain the processed feature sequence <inline-formula><mml:math id="M11"><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula>:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>S</mml:mi><mml:mn>6</mml:mn><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>S</italic>6 denotes the selective scanning state space model&#x00027;s operation on the feature sequence.</p>
<p>After processing all the directional feature sequences, SS2D merges the sequences <inline-formula><mml:math id="M13"><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula> to reconstruct the 2D feature representation:</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">merge</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the function merge denotes the fusion of the features from the four directions to form the final 2D feature representation.</p>
<p>By scanning from four different directions, processing the feature sequences, and merging them, SS2D effectively captures the global spatial information of the image, thereby enhancing the model&#x00027;s perception and understanding of visual tasks. The resulting feature <inline-formula><mml:math id="M15"><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:mrow></mml:math></inline-formula> contains rich contextual relationships, providing a comprehensive and efficient representation for subsequent visual analysis tasks.</p>
</sec>
<sec>
<title>3.2 Model structure</title>
<p>This paper proposes a network structure called MUNet, which combines SSM and the encoder-decoder architecture of UNet for efficient image segmentation. As shown in <xref ref-type="fig" rid="F1">Figure 1A</xref>, MUNet first partitions the image into patches and performs linear embedding to obtain initial feature representations. These features are then processed by multiple SD-SSM block, where each module scans the feature sequences from different directions, capturing the global contextual information of the image for feature modeling and enhancement. The encoder gradually compresses the feature map to extract multi-scale information, while the Skip Connection passes features from the encoding process directly to the decoder to preserve image details. In the decoder, the network progressively restores the spatial resolution of the feature map and achieves accurate segmentation by integrating the multi-level features from the encoder. Patch Merging and Patch Expanding layers are used for feature compression and expansion, ensuring smooth information flow throughout the encoding-decoding process. Finally, through the combination of multiple SD-SSM block and skip connections, MUNet effectively captures both global and local features of the image, enhancing segmentation accuracy and efficiency.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>MUNet Architecture. <bold>(A)</bold> Illustrates the overall structure of MUNet, which follows an encoder-decoder design. The input image is processed through linear embedding, SD-SSM blocks, and multiple rounds of patch merging, with skip connections between the encoder and decoder layers facilitating information transfer. <bold>(B)</bold> Provides a detailed view of the internal structure of the SD-SSM block.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-19-1513059-g0001.tif"/>
</fig>
</sec>
<sec>
<title>3.3 SD-SSM block</title>
<p>The SD-SSM Block is a key module in the MUNet network, as shown in <xref ref-type="fig" rid="F1">Figure 1B</xref>. Its structure comprises two main branches: <italic>X</italic><sub>1</sub> and <italic>X</italic><sub>2</sub>. The <italic>X</italic><sub>1</sub> branch first applies batch normalization to the input features and then captures multi-scale information through multiple layers of dilated convolution using the SD-Conv module. The <italic>X</italic><sub>2</sub> branch normalizes the features through layer normalization and linear transformation, combined with the SiLU activation function to enhance non-linear representation capability. Subsequently, the features are processed through SD-Conv and SS2D modules to ensure the capture of global context information. The features from both branches are finally fused and connected to the residuals of the input features to enhance representational capacity.</p>
<p>The core module of the SD-SSM Block, SD-Conv, is composed of SCConv (Spatial and Channel Reconstruction Convolution) (Li et al., <xref ref-type="bibr" rid="B22">2023a</xref>) and DW Conv (Depthwise Separable Convolution) (Huang et al., <xref ref-type="bibr" rid="B18">2023</xref>), as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. SCConv reconstructs spatial and channel information through two units: the Spatial Reconstruction Unit (SRU), which suppresses spatial redundancy through a separated reconstruction method, and the Channel Reconstruction Unit (CRU), which eliminates channel redundancy through a split-transform-merge strategy. By integrating SCConv and DW Conv, SD-Conv effectively compresses spatial and channel redundancies among features, forming an efficient convolutional module. This module reduces redundant computations while preserving the representational capacity of the model, enabling better learning of key features in the image, particularly those in tumors, and enhancing the model&#x00027;s segmentation performance.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The architecture of SCConv integrated with the Spatial Reconstruction Unit (SRU) and the Channel Reconstruction Unit (CRU).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-19-1513059-g0002.tif"/>
</fig>
</sec>
<sec>
<title>3.4 Skip connetions</title>
<p>Two SD-SSM Blocks are used in MUNet&#x00027;s encoder and decoder to effectively model both local and global features of the image. Each level of the encoder and decoder employs skip connections to mix multi-scale features with the upsampled output, enhancing spatial details by merging shallow and deep features. These skip connections ensure that high-resolution features from earlier layers of the encoder are preserved and fully utilized in the decoding process, maintaining crucial spatial information throughout the network and improving segmentation accuracy. This design enables MUNet to capture fine-grained details and contextual information simultaneously, achieving more precise and robust segmentation results, particularly in complex visual scenarios like tumor boundaries.</p>
</sec>
<sec>
<title>3.5 Loss function</title>
<p>In the domain of MRI brain tumor segmentation, the key evaluation metrics are the overlap between the segmentation results and the ground truth, as well as the accuracy and similarity of the boundaries. To address these metrics effectively, we design a weighted loss function that combines mIoU Loss, Dice Loss, and Boundary Loss. Each loss function optimizes a different aspect of the segmentation task, ensuring that the network achieves comprehensive and balanced performance.</p>
<p>The mean Intersection over Union (mIoU) loss is used to optimize the overlap between the predicted segmentation <italic>P</italic> and the ground truth <italic>G</italic>. It is defined as:</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">mIoU</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02229;</mml:mo><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0222A;</mml:mo><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>N</italic> is the total number of pixels in the image, <italic>P</italic><sub><italic>i</italic></sub> is the set of pixels predicted as part of the tumor in the <italic>i</italic>-th class, and <italic>G</italic><sub><italic>i</italic></sub> is the set of ground truth pixels for the tumor in the <italic>i</italic>-th class. The mIoU loss penalizes regions where the segmentation and ground truth do not overlap well, focusing on improving the overall overlap accuracy.</p>
<p>The Dice loss aims to maximize the similarity between the predicted segmentation and the ground truth. It is defined as:</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Dice</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mstyle displaystyle="false"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mstyle displaystyle="false"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:mstyle displaystyle="false"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>P</italic><sub><italic>i</italic></sub> and <italic>G</italic><sub><italic>i</italic></sub> are the prediction and ground truth for each pixel. The Dice loss emphasizes the correct classification of tumor regions, focusing on the balance between false positives and false negatives, thus optimizing both sensitivity and precision.</p>
<p>The Boundary loss is used to refine the accuracy of the segmentation boundaries, which is crucial for brain tumor segmentation. It is defined as:</p>
<disp-formula id="E14"><label>(14)</label><mml:math id="M18"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Boundary</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="false"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">boundary</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>d</italic><sub>boundary</sub>(<italic>P</italic><sub><italic>i</italic></sub>, <italic>G</italic><sub><italic>i</italic></sub>) is the distance between the predicted boundary and the ground truth boundary for pixel <italic>i</italic>, and <italic>N</italic> is the total number of pixels in the boundary region. The Boundary loss optimizes the fine details of the segmentation, ensuring that the predicted boundaries closely match the ground truth boundaries.</p>
<p>The final loss function is a weighted combination of the above three losses:</p>
<disp-formula id="E15"><label>(15)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">total</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">mIoU</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B2;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Dice</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">Boundary</mml:mtext></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B1;, &#x003B2;, &#x003B3; are the weights that control the contribution of each loss term.</p>
<p>By combining these three loss components, the total loss function optimizes the segmentation task from multiple perspectives: enhancing overall overlap accuracy (mIoU), improving the similarity between predicted and actual tumor regions (Dice), and refining boundary precision (Boundary). This comprehensive approach allows for more accurate and effective segmentation results in MRI brain tumor analysis.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Experiments</title>
<sec>
<title>4.1 Experimental setup</title>
<p><bold>Datasets</bold> This study utilizes three main datasets for experiments: the BraTS2020 dataset, the BraTS2018 dataset (Bakas et al., <xref ref-type="bibr" rid="B5">2017</xref>, <xref ref-type="bibr" rid="B6">2018</xref>; Menze et al., <xref ref-type="bibr" rid="B26">2014</xref>), and an independently validated LGG segmentation dataset (Buda et al., <xref ref-type="bibr" rid="B8">2019</xref>). The LGG segmentation dataset is sourced from The Cancer Imaging Archive (TCIA) and includes MRI images from 110 patients in the Cancer Genome Atlas (TCGA) Low-Grade Glioma (LGG) collection, with a total of 3,929 images. These images are used for research on low-grade glioma segmentation, with the training set containing 2,750 images and the test set containing 1,179 images. <xref ref-type="fig" rid="F3">Figure 3A</xref> shows an example from this dataset. The BraTS2020 dataset focuses on brain tumor segmentation, particularly for evaluating advanced methods in tumor segmentation using multimodal MRI scans. The BraTS2020 dataset provides a large training set of 369 MRI scan images and a validation set of 125 scans. Similarly, the BraTS2018 dataset is used for brain tumor segmentation, containing 285 training images and 66 validation images. Each MRI scan has a size of 240 &#x000D7; 240 &#x000D7; 155, with each case including multiple modalities such as T1, T1c, T2, and FLAIR. <xref ref-type="fig" rid="F3">Figure 3B</xref> shows an example from the BraTS dataset.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Dataset sample display. <bold>(A)</bold> LGG segmentation dataset display. <bold>(B)</bold> BraTS dataset display.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-19-1513059-g0003.tif"/>
</fig>
<p><bold>Experimental environment</bold> The experiment was conducted on a high-performance server with the following hardware configuration: Intel Xeon Gold 6226R processor, NVIDIA Tesla V100 GPU (32GB memory), 128GB DDR4 RAM, and 1TB NVMe SSD storage, running Ubuntu 20.04 LTS as the operating system. For the software environment, PyTorch 1.10 was used as the deep learning framework, with CUDA 11.4 and cuDNN 8.2 for acceleration, and Python 3.8 as the programming language. The scientific computing libraries included NumPy 1.21 and SciPy 1.7, while Pandas 1.3 and Matplotlib 3.4 were used for data processing and visualization. Additionally, OpenCV 4.5 was installed for image processing, and scikit-learn 0.24 for data analysis and model evaluation. All software packages were managed using the conda environment management tool to ensure reproducibility and compatibility of dependencies.</p>
<p><bold>Evaluation metrics</bold> The evaluation metrics used in this paper include Kappa (k), Dice Similarity Coefficient (DSC), Intersection over Union (IoU), Sensitivity (S), Precision (P), Specificity (Sp), Accuracy (A), and Balanced Accuracy (BA).</p>
<p>Kappa is used to measure the agreement between the predicted and actual classifications, considering the possibility of the agreement occurring by chance.</p>
<disp-formula id="E16"><label>(16)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where: <italic>P</italic><sub><italic>o</italic></sub> is the observed agreement (the proportion of correct predictions). <italic>P</italic><sub><italic>e</italic></sub> is the expected agreement (the proportion of correct predictions by chance).</p>
<p>DSC is used to measure the degree of overlap between two sets, while IoU measures the ratio of the intersection and the union between the predicted and the ground truth regions.</p>
<disp-formula id="E17"><label>(17)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>D</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mo>|</mml:mo><mml:mi>X</mml:mi><mml:mo>&#x02229;</mml:mo><mml:mi>Y</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>X</mml:mi><mml:mo>|</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>|</mml:mo><mml:mi>Y</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E18"><label>(18)</label><mml:math id="M22"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:mi>X</mml:mi><mml:mo>&#x02229;</mml:mo><mml:mi>Y</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>X</mml:mi><mml:mo>&#x0222A;</mml:mo><mml:mi>Y</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>X</italic> and <italic>Y</italic> represent the sets of predicted and ground truth pixels, respectively.</p>
<p>Sensitivity (S) measures the model&#x00027;s ability to correctly identify positive instances, precision (P) represents the correctness of the predicted positive instances, specificity (Sp) measures the ability to correctly identify negative instances, and accuracy (A) measures the overall correctness of the model&#x00027;s predictions.</p>
<disp-formula id="E19"><label>(19)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E20"><label>(20)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E21"><label>(21)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>S</mml:mi><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E22"><label>(22)</label><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>TP</italic> stands for true positives, <italic>TN</italic> stands for true negatives, <italic>FP</italic> stands for false positives, and <italic>FN</italic> stands for false negatives.</p>
<p>Balanced Accuracy is used to account for imbalanced data, providing the average of sensitivity and specificity.</p>
<disp-formula id="E23"><label>(23)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>B</mml:mi><mml:mi>A</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>S</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>S</mml:mi><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Hausdorff95 is a metric commonly used in segmentation tasks to measure the spatial distance between the predicted boundary and the ground truth boundary. Unlike the traditional Hausdorff Distance, which considers the maximum distance between two point sets, Hausdorff95 focuses on the 95th percentile distance, effectively reducing the influence of outliers and providing a more robust evaluation for medical image segmentation tasks where extreme outliers might distort the overall assessment.</p>
<disp-formula id="E24"><label>(24)</label><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mn>95</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>B</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo class="qopname">max</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mn>95</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>B</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mn>95</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>A</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where, <italic>h</italic><sub>95</sub>(<italic>A, B</italic>) represents the 95th percentile of the set of minimum distances from points in set <italic>A</italic> to the closest points in set <italic>B</italic>. <italic>A</italic> and <italic>B</italic> are typically the sets of points that represent the boundaries of the ground truth segmentation and the predicted segmentation, respectively. This measure ensures that 95% of the points on the predicted boundary are within a certain distance from the ground truth boundary, making it a more resilient metric for segmentation accuracy, particularly in scenarios where small boundary discrepancies are permissible.</p>
</sec>
<sec>
<title>4.2 Results</title>
<p>In this paper, the MUNet model demonstrated outstanding performance on both the BraTS2020 and BraTS2018 datasets, significantly surpassing other existing models. As shown in <xref ref-type="table" rid="T1">Table 1</xref>, MUNet&#x00027;s performance on the three key metrics-enhancing tumor (ET), whole tumor (WT), and tumor core (TC)-is highlighted.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Comparison of different methods using DSC and Hausdorff95 metrics for BraTS2020 dataset and BraTS2018 dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center" colspan="3"><bold>DSC (BraTS2020)</bold></th>
<th valign="top" align="center" colspan="3"><bold>Hausdorff95 (BraTS2020)</bold></th>
<th valign="top" align="center" colspan="3"><bold>DSC (BraTS2018)</bold></th>
<th valign="top" align="center" colspan="3"><bold>Hausdorff95 (BraTS2018)</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>U-Net</bold></td>
<td valign="top" align="center">0.783</td>
<td valign="top" align="center">0.882</td>
<td valign="top" align="center">0.801</td>
<td valign="top" align="center">5.835</td>
<td valign="top" align="center">5.447</td>
<td valign="top" align="center">7.123</td>
<td valign="top" align="center">0.724</td>
<td valign="top" align="center">0.851</td>
<td valign="top" align="center">0.747</td>
<td valign="top" align="center">6.554</td>
<td valign="top" align="center">12.645</td>
<td valign="top" align="center">11.035</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Attention U-Net</bold> (Oktay et al., <xref ref-type="bibr" rid="B29">2018</xref>)</td>
<td valign="top" align="center">0.775</td>
<td valign="top" align="center">0.859</td>
<td valign="top" align="center">0.798</td>
<td valign="top" align="center">4.586</td>
<td valign="top" align="center">5.253</td>
<td valign="top" align="center">7.174</td>
<td valign="top" align="center">0.765</td>
<td valign="top" align="center">0.869</td>
<td valign="top" align="center">0.758</td>
<td valign="top" align="center">7.596</td>
<td valign="top" align="center">10.253</td>
<td valign="top" align="center">11.174</td>
</tr>
<tr>
<td valign="top" align="left"><bold>ResU-Net</bold> (Maji et al., <xref ref-type="bibr" rid="B25">2022</xref>)</td>
<td valign="top" align="center">0.812</td>
<td valign="top" align="center">0.895</td>
<td valign="top" align="center">0.813</td>
<td valign="top" align="center">4.759</td>
<td valign="top" align="center">5.789</td>
<td valign="top" align="center">6.597</td>
<td valign="top" align="center">0.795</td>
<td valign="top" align="center">0.892</td>
<td valign="top" align="center">0.758</td>
<td valign="top" align="center">6.993</td>
<td valign="top" align="center">8.597</td>
<td valign="top" align="center">10.046</td>
</tr>
<tr>
<td valign="top" align="left"><bold>FE-HU-NET</bold> (Nizamani et al., <xref ref-type="bibr" rid="B28">2023</xref>)</td>
<td valign="top" align="center">0.802</td>
<td valign="top" align="center">0.870</td>
<td valign="top" align="center">0.815</td>
<td valign="top" align="center">4.859</td>
<td valign="top" align="center">5.253</td>
<td valign="top" align="center">5.764</td>
<td valign="top" align="center">0.742</td>
<td valign="top" align="center">0.872</td>
<td valign="top" align="center">0.745</td>
<td valign="top" align="center">5.894</td>
<td valign="top" align="center">8.243</td>
<td valign="top" align="center">9.765</td>
</tr>
<tr>
<td valign="top" align="left"><bold>HC-Mamba</bold> (Xu, <xref ref-type="bibr" rid="B39">2024</xref>)</td>
<td valign="top" align="center">0.795</td>
<td valign="top" align="center">0.899</td>
<td valign="top" align="center">0.812</td>
<td valign="top" align="center">5.358</td>
<td valign="top" align="center">4.125</td>
<td valign="top" align="center">7.766</td>
<td valign="top" align="center">0.812</td>
<td valign="top" align="center">0.883</td>
<td valign="top" align="center">0.787</td>
<td valign="top" align="center">5.459</td>
<td valign="top" align="center">6.248</td>
<td valign="top" align="center">9.764</td>
</tr>
<tr>
<td valign="top" align="left"><bold>Mamba-UNet</bold> (Wang et al., <xref ref-type="bibr" rid="B37">2024</xref>)</td>
<td valign="top" align="center">0.813</td>
<td valign="top" align="center">0.902</td>
<td valign="top" align="center">0.798</td>
<td valign="top" align="center">3.389</td>
<td valign="top" align="center">4.243</td>
<td valign="top" align="center">6.766</td>
<td valign="top" align="center">0.802</td>
<td valign="top" align="center">0.906</td>
<td valign="top" align="center">0.801</td>
<td valign="top" align="center">5.127</td>
<td valign="top" align="center">6.873</td>
<td valign="top" align="center">6.766</td>
</tr>
<tr>
<td valign="top" align="left"><bold>SwimUNet</bold> (Cao et al., <xref ref-type="bibr" rid="B9">2022</xref>)</td>
<td valign="top" align="center">0.792</td>
<td valign="top" align="center">0.892</td>
<td valign="top" align="center">0.745</td>
<td valign="top" align="center">5.347</td>
<td valign="top" align="center">6.729</td>
<td valign="top" align="center">8.588</td>
<td valign="top" align="center">0.776</td>
<td valign="top" align="center">0.847</td>
<td valign="top" align="center">0.759</td>
<td valign="top" align="center">6.798</td>
<td valign="top" align="center">8.257</td>
<td valign="top" align="center">8.712</td>
</tr>
<tr>
<td valign="top" align="left"><bold>TransUNet</bold> (Chen et al., <xref ref-type="bibr" rid="B11">2021</xref>)</td>
<td valign="top" align="center">0.798</td>
<td valign="top" align="center">0.877</td>
<td valign="top" align="center">0.712</td>
<td valign="top" align="center">3.598</td>
<td valign="top" align="center">5.871</td>
<td valign="top" align="center">6.766</td>
<td valign="top" align="center">0.762</td>
<td valign="top" align="center">0.850</td>
<td valign="top" align="center">0.765</td>
<td valign="top" align="center">6.389</td>
<td valign="top" align="center">8.247</td>
<td valign="top" align="center">8.153</td>
</tr>
<tr>
<td valign="top" align="left"><bold>MUNet</bold></td>
<td valign="top" align="center"><bold>0.835</bold></td>
<td valign="top" align="center"><bold>0.915</bold></td>
<td valign="top" align="center"><bold>0.823</bold></td>
<td valign="top" align="center"><bold>2.421</bold></td>
<td valign="top" align="center"><bold>3.755</bold></td>
<td valign="top" align="center">6.437</td>
<td valign="top" align="center"><bold>0.815</bold></td>
<td valign="top" align="center"><bold>0.901</bold></td>
<td valign="top" align="center"><bold>0.815</bold></td>
<td valign="top" align="center"><bold>4.389</bold></td>
<td valign="top" align="center"><bold>6.243</bold></td>
<td valign="top" align="center"><bold>6.152</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The optimal results are indicated in bold.</p>
</table-wrap-foot>
</table-wrap>
<p>On the BraTS2020 dataset, MUNet exhibited a notable improvement over other methods. Specifically, for ET segmentation, MUNet achieved a DSC score of 0.835, approximately 6.6% higher than the traditional U-Net. For WT segmentation, MUNet reached a DSC score of 0.915, outperforming ResU-Net by about 2.2%. Additionally, MUNet showed clear advantages in TC segmentation, improving by around 2.7% compared to U-Net. In terms of boundary accuracy, MUNet also performed exceptionally well, with the Hausdorff95 distance for ET reduced by nearly 58% compared to U-Net, indicating significant improvements in boundary capturing and morphological recognition.</p>
<p>On the BraTS2018 dataset, MUNet also exhibited substantial improvements. For ET segmentation, MUNet&#x00027;s DSC score increased by about 12.5% compared to traditional U-Net. WT segmentation accuracy improved by around 5.9%, meaning MUNet can more effectively capture the global morphological characteristics of tumors. Notably, MUNet also showed significant improvements in Hausdorff95 distance, reducing the ET score by approximately 37% compared to other models.</p>
<p>This result also indicates that, compared to existing CNN-based and Transformer-based models, MUNet demonstrates superior performance in tumor segmentation tasks. For example, although ResU-Net improves feature propagation through residual connections, it still falls short when handling complex MRI images. MUNet, by integrating both global and local features and leveraging the SD-SSM module to effectively capture detailed information, significantly improves segmentation accuracy. In comparison to SwimUNet, MUNet also shows a clear advantage in detail processing. While SwimUNet enhances global context information, it does not perform as well as MUNet in the reconstruction of fine boundaries. MUNet ensures precise boundary modeling through skip connections and multi-layer SD-Conv modules, resulting in a lower Hausdorff95 distance and higher segmentation accuracy.</p>
<p><xref ref-type="fig" rid="F4">Figure 4</xref> visualizes the tumor detection results of MUNet for brain tumors. As seen in the figure, MUNet demonstrates remarkable accuracy in segmenting tumor regions, particularly excelling in delineating boundary areas, where it clearly outperforms traditional models. The segmentation results not only provide a clear depiction of the tumor&#x00027;s morphological features but also retain critical detailed information. Additionally, MUNet exhibits excellent global consistency in segmenting both WT and TC, with the tumor contours and core regions represented comprehensively and accurately. These results highlight MUNet&#x00027;s strong capability in handling complex brain tumor shapes and its potential for precise tumor analysis in medical imaging.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Display of MUNet segmentation performance on the BraTS dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-19-1513059-g0004.tif"/>
</fig>
<p><bold>Computational complexity analysis</bold> As shown in <xref ref-type="table" rid="T1">Table 1</xref>, we further compared the computational complexity of MUNet with other existing models, specifically considering the number of floating-point operations (FLOPs) and the number of model parameters. The data in the table clearly show that MUNet has a significant advantage in both FLOPs and parameter count. Firstly, MUNet has 140.97 GFLOPs, which is notably lower than most traditional models, especially SwimUNet and TransUNet, which have 370.31 GFLOPs and 390.76 GFLOPs, respectively. This indicates that MUNet is more computationally efficient and can perform the same tasks with fewer computational resources, reducing both computational cost and runtime. Additionally, MUNet has 7.27M parameters, which is significantly fewer compared to models such as ResU-Net (25.75M), SwimUNet (25.18M), and TransUNet (20.45M). The smaller parameter count not only helps accelerate model training but also reduces memory usage, facilitating efficient deployment even under hardware constraints.</p>
</sec>
<sec>
<title>4.3 Ablation study</title>
<p>In the ablation experiments of this paper, we conducted a detailed analysis of MUNet&#x00027;s performance on the BraTS2020 and BraTS2018 datasets by progressively adding each of its modules. Additionally, we analyzed the impact of the proposed loss functions on MUNet&#x00027;s performance.</p>
<sec>
<title>4.3.1 Ablation experiments between components</title>
<p><bold>Performance on the BraTS2020 dataset</bold> As shown in <xref ref-type="table" rid="T1">Table 1</xref>, the baseline U-Net model achieved DSC scores of 0.783, 0.882, and 0.801 for ET, WT, and TC segmentation, respectively, with Hausdorff95 distances of 5.835, 5.447, and 7.123. With the introduction of the SD-SSM Block, the model&#x00027;s ability to capture global and local features improved, resulting in increased DSC scores of 0.795, 0.895, and 0.805 for ET, WT, and TC, along with a reduction in Hausdorff95 distances. Notably, the Hausdorff95 distance for WT decreased from 5.447 to 4.358, indicating a significant improvement in boundary accuracy. When the SD-Conv structure was further added, the model&#x00027;s performance improved again, with DSC scores increasing to 0.815, 0.899, and 0.810 for ET, WT, and TC, respectively, and a reduction in Hausdorff95 distances across all metrics, especially for ET, which decreased to 4.524. This improvement suggests that the SD-Conv structure helped reduce feature redundancy and enhance computational efficiency, contributing to better segmentation accuracy while maintaining high-resolution features. Finally, after the introduction of the newly designed loss function, the model reached optimal performance. The DSC scores for ET, WT, and TC rose to 0.835, 0.915, and 0.823, respectively, significantly outperforming all previous combinations. Notably, the Hausdorff95 distance for ET decreased to 2.421, a reduction of approximately 58%. This demonstrates that the new loss function played a crucial role in optimizing the overlap, similarity, and boundary accuracy of the segmentation results, significantly improving the model&#x00027;s ability to handle complex tumor morphologies. From these ablation experiment results, it can be seen that the SD-SSM Block and SD-Conv modules work synergistically, not only improving the model&#x00027;s ability to capture global and local features but also enhancing efficiency by reducing feature redundancy. Moreover, the new loss function played a critical role in improving boundary accuracy and preserving fine details. Together, these components have enabled MUNet to demonstrate significant advantages in brain tumor segmentation tasks.</p>
<p><bold>Performance on the BraTS2018 dataset</bold> In the ablation experiments on the BraTS2018 dataset, the results followed a similar trend to those observed in the BraTS2020 dataset. As shown in <xref ref-type="table" rid="T2">Table 2</xref>, the baseline U-Net model achieved DSC scores of 0.724, 0.851, and 0.747 for ET, WT, and TC segmentation, respectively, with corresponding Hausdorff95 distances of 6.554, 12.645, and 11.035, serving as the baseline performance. When the SD-SSM Block was added, although the DSC for WT slightly decreased to 0.779, the DSC for ET and TC improved slightly to 0.764 and 0.728, respectively. In addition, there was an improvement in Hausdorff95 distances, particularly for WT, where the distance decreased from 12.645 to 11.273, indicating some enhancement in boundary accuracy. With the introduction of the SD-Conv module, the overall performance of the model improved significantly. The DSC scores for ET, WT, and TC increased to 0.796, 0.885, and 0.787, respectively, and the Hausdorff95 distances dropped considerably, especially for WT, where it decreased from 11.273 to 8.597. This suggests that the SD-Conv module greatly contributed to feature extraction and boundary handling. Finally, when the new loss function was introduced, the model reached its optimal performance. The DSC scores for ET, WT, and TC rose to 0.815, 0.901, and 0.815, respectively, and the Hausdorff95 distances significantly decreased across all metrics, particularly for ET, where the distance dropped to 4.389. This indicates that the new loss function greatly enhanced the model&#x00027;s segmentation accuracy, especially in handling boundaries and complex tumor morphologies.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Comparison and analysis of dataset model efficiency.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Methods</bold></th>
<th valign="top" align="center"><bold>FLOPs (GFLOPs)</bold></th>
<th valign="top" align="center"><bold>Number of parameters (Millions)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">U-Net</td>
<td valign="top" align="center">142.05</td>
<td valign="top" align="center">6.53M</td>
</tr>
<tr>
<td valign="top" align="left">Attention U-Net</td>
<td valign="top" align="center">156.21</td>
<td valign="top" align="center">7.49M</td>
</tr>
<tr>
<td valign="top" align="left">ResU-Net</td>
<td valign="top" align="center">242.96</td>
<td valign="top" align="center">25.75M</td>
</tr>
<tr>
<td valign="top" align="left">FE-HU-NET</td>
<td valign="top" align="center">241.97</td>
<td valign="top" align="center">11.75M</td>
</tr>
<tr>
<td valign="top" align="left">HC-Mamba</td>
<td valign="top" align="center">169.25</td>
<td valign="top" align="center">9.38M</td>
</tr>
<tr>
<td valign="top" align="left">Mamba-UNet</td>
<td valign="top" align="center">199.89</td>
<td valign="top" align="center">22.65M</td>
</tr>
<tr>
<td valign="top" align="left">SwimUNet</td>
<td valign="top" align="center">370.31</td>
<td valign="top" align="center">25.18M</td>
</tr>
<tr>
<td valign="top" align="left">TransUNet</td>
<td valign="top" align="center">390.76</td>
<td valign="top" align="center">20.45M</td>
</tr>
<tr>
<td valign="top" align="left">MUNet</td>
<td valign="top" align="center"><bold>140.97</bold></td>
<td valign="top" align="center">7.27M</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The optimal results are indicated in bold.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.3.2 Ablation experiments on the loss function</title>
<p><bold>Performance on the BraTS2020 dataset</bold> As shown in <xref ref-type="table" rid="T5">Table 5</xref>, the complete MUNet model achieved a DSC score of 0.835 for ET segmentation. However, when the mIoU loss function was removed, the DSC significantly dropped by around 20%, down to 0.665. At the same time, the Hausdorff95 distance increased substantially from 2.421 to 6.586, almost a 2.7-fold increase, indicating the critical role of mIoU in enhancing overall segmentation accuracy. When the Dice loss was removed, the DSC for ET decreased to 0.709, about 15% lower than the complete model, and the Hausdorff95 distance increased to 8.687, highlighting the importance of Dice loss in improving regional similarity. Similarly, removing the Boundary loss resulted in a DSC drop to 0.642, a reduction of approximately 23%, while the Hausdorff95 distance increased to 9.126, nearly quadrupling, emphasizing the significance of Boundary loss in ensuring segmentation boundary accuracy. For WT and TC segmentation, similar trends were observed. Removing the mIoU loss led to a roughly 25% decrease in WT&#x00027;s DSC, and the Hausdorff95 distance more than doubled. Removing Dice and Boundary losses resulted in approximately 20% reductions in segmentation accuracy, with significant increases in Hausdorff95 distance, indicating that each loss function plays a key role in optimizing performance for different segmentation targets.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Ablation experiments on different combinations of MUNet on the BraTS2020 dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center" colspan="3"><bold>DSC</bold></th>
<th valign="top" align="center" colspan="3"><bold>Hausdorff95</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>U-Net</bold></td>
<td valign="top" align="center">0.783</td>
<td valign="top" align="center">0.882</td>
<td valign="top" align="center">0.801</td>
<td valign="top" align="center">5.835</td>
<td valign="top" align="center">5.447</td>
<td valign="top" align="center">7.123</td>
</tr>
<tr>
<td valign="top" align="left"><bold>U-Net &#x0002B; SD-SSM Block</bold></td>
<td valign="top" align="center">0.795</td>
<td valign="top" align="center">0.895</td>
<td valign="top" align="center">0.805</td>
<td valign="top" align="center">5.105</td>
<td valign="top" align="center">4.358</td>
<td valign="top" align="center">6.578</td>
</tr>
<tr>
<td valign="top" align="left"><bold>U-Net &#x0002B; SD-SSM Block &#x0002B; SD-Conv</bold></td>
<td valign="top" align="center">0.815</td>
<td valign="top" align="center">0.899</td>
<td valign="top" align="center">0.810</td>
<td valign="top" align="center">4.524</td>
<td valign="top" align="center">3.698</td>
<td valign="top" align="center">6.581</td>
</tr>
<tr>
<td valign="top" align="left"><bold>U-Net &#x0002B; SD-SSM Block &#x0002B; SD-Conv &#x0002B; Loss function</bold></td>
<td valign="top" align="center"><bold>0.835</bold></td>
<td valign="top" align="center"><bold>0.915</bold></td>
<td valign="top" align="center"><bold>0.823</bold></td>
<td valign="top" align="center"><bold>2.421</bold></td>
<td valign="top" align="center"><bold>3.755</bold></td>
<td valign="top" align="center"><bold>6.437</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The optimal results are indicated in bold.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Ablation experiments on different combinations of MUNet on the BraTS2018 dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center" colspan="3"><bold>DSC</bold></th>
<th valign="top" align="center" colspan="3"><bold>Hausdorff95</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>U-Net</bold></td>
<td valign="top" align="center">0.724</td>
<td valign="top" align="center">0.851</td>
<td valign="top" align="center">0.747</td>
<td valign="top" align="center">6.554</td>
<td valign="top" align="center">12.645</td>
<td valign="top" align="center">11.035</td>
</tr>
<tr>
<td valign="top" align="left"><bold>U-Net &#x0002B; SD-SSM Block</bold></td>
<td valign="top" align="center">0.764</td>
<td valign="top" align="center">0.779</td>
<td valign="top" align="center">0.728</td>
<td valign="top" align="center">6.586</td>
<td valign="top" align="center">11.273</td>
<td valign="top" align="center">10.175</td>
</tr>
<tr>
<td valign="top" align="left"><bold>U-Net &#x0002B; SD-SSM Block &#x0002B; SD-Conv</bold></td>
<td valign="top" align="center">0.796</td>
<td valign="top" align="center">0.885</td>
<td valign="top" align="center">0.787</td>
<td valign="top" align="center">5.993</td>
<td valign="top" align="center">8.597</td>
<td valign="top" align="center">7.046</td>
</tr>
<tr>
<td valign="top" align="left"><bold>U-Net &#x0002B; SD-SSM Block &#x0002B; SD-Conv &#x0002B; Loss function</bold></td>
<td valign="top" align="center"><bold>0.815</bold></td>
<td valign="top" align="center"><bold>0.901</bold></td>
<td valign="top" align="center"><bold>0.815</bold></td>
<td valign="top" align="center"><bold>4.389</bold></td>
<td valign="top" align="center"><bold>6.243</bold></td>
<td valign="top" align="center"><bold>6.152</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The optimal results are indicated in bold.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Ablation experiments on the loss function of MUNet.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center" colspan="3"><bold>DSC (BraTS2020)</bold></th>
<th valign="top" align="center" colspan="3"><bold>Hausdorff95 (BraTS2020)</bold></th>
<th valign="top" align="center" colspan="3"><bold>DSC (BraTS2018)</bold></th>
<th valign="top" align="center" colspan="3"><bold>Hausdorff95 (BraTS2018)</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
<th valign="top" align="center"><bold>ET</bold></th>
<th valign="top" align="center"><bold>WT</bold></th>
<th valign="top" align="center"><bold>TC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>MUNet</bold></td>
<td valign="top" align="center"><bold>0.835</bold></td>
<td valign="top" align="center"><bold>0.915</bold></td>
<td valign="top" align="center"><bold>0.823</bold></td>
<td valign="top" align="center"><bold>2.421</bold></td>
<td valign="top" align="center"><bold>3.755</bold></td>
<td valign="top" align="center"><bold>6.437</bold></td>
<td valign="top" align="center"><bold>0.815</bold></td>
<td valign="top" align="center"><bold>0.901</bold></td>
<td valign="top" align="center"><bold>0.815</bold></td>
<td valign="top" align="center"><bold>4.389</bold></td>
<td valign="top" align="center"><bold>6.243</bold></td>
<td valign="top" align="center"><bold>6.152</bold></td>
</tr>
<tr>
<td valign="top" align="left"><bold>w/o</bold> <inline-formula><mml:math id="M29"><mml:mrow><mml:msub><mml:mi>&#x02112;</mml:mi><mml:mrow><mml:mtext>mIoU</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">0.665</td>
<td valign="top" align="center">0.689</td>
<td valign="top" align="center">0.728</td>
<td valign="top" align="center">6.586</td>
<td valign="top" align="center">11.253</td>
<td valign="top" align="center">15.174</td>
<td valign="top" align="center">0.710</td>
<td valign="top" align="center">0.724</td>
<td valign="top" align="center">0.763</td>
<td valign="top" align="center">7.258</td>
<td valign="top" align="center">15.254</td>
<td valign="top" align="center">14.184</td>
</tr>
<tr>
<td valign="top" align="left"><bold>w/o</bold> <inline-formula><mml:math id="M30"><mml:mrow><mml:msub><mml:mi>&#x02112;</mml:mi><mml:mrow><mml:mtext>Dice</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">0.709</td>
<td valign="top" align="center">0.714</td>
<td valign="top" align="center">0.708</td>
<td valign="top" align="center">8.687</td>
<td valign="top" align="center">10.543</td>
<td valign="top" align="center">11.258</td>
<td valign="top" align="center">0.684</td>
<td valign="top" align="center">0.692</td>
<td valign="top" align="center">0.719</td>
<td valign="top" align="center">8.573</td>
<td valign="top" align="center">12.574</td>
<td valign="top" align="center">13.245</td>
</tr>
<tr>
<td valign="top" align="left"><bold>w/o</bold> <inline-formula><mml:math id="M31"><mml:mrow><mml:msub><mml:mi>&#x02112;</mml:mi><mml:mrow><mml:mtext>Boundary</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center">0.642</td>
<td valign="top" align="center">0.671</td>
<td valign="top" align="center">0.665</td>
<td valign="top" align="center">9.126</td>
<td valign="top" align="center">10.374</td>
<td valign="top" align="center">12.149</td>
<td valign="top" align="center">0.601</td>
<td valign="top" align="center">0.679</td>
<td valign="top" align="center">0.712</td>
<td valign="top" align="center">10.589</td>
<td valign="top" align="center">14.153</td>
<td valign="top" align="center">15.766</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The optimal results are indicated in bold.</p>
</table-wrap-foot>
</table-wrap>
<p><bold>Performance on the BraTS2018 dataset</bold> For the BraTS2018 dataset, the complete MUNet model achieved a DSC score of 0.815 for ET segmentation. After removing the mIoU loss, the DSC dropped by approximately 13%, to 0.710, while the Hausdorff95 distance increased by 65%, from 4.389 to 7.258. This shows that mIoU is equally crucial for improving global segmentation performance on this dataset. Removing the Dice loss reduced the DSC for ET to 0.684, a decrease of around 16%, and the Hausdorff95 distance increased to 8.573, nearly doubling. Similarly, removing the Boundary loss led to a 26% reduction in ET&#x00027;s DSC, while the Hausdorff95 distance increased to 10.589, almost 2.4 times higher, demonstrating the essential role of Boundary loss in capturing accurate boundaries in complex tumor morphologies. The segmentation for WT and TC also showed similar declines. Removing the mIoU loss resulted in a 20% reduction in WT&#x00027;s DSC, while the Hausdorff95 distance roughly doubled. Removing Dice and Boundary losses decreased segmentation accuracy by about 20%, with significant increases in Hausdorff95 distance, confirming the multi-dimensional improvement in segmentation performance provided by these loss functions.</p>
</sec>
</sec>
<sec>
<title>4.4 Independent validation</title>
<p>The independent validation on the LGG segmentation dataset, as shown in <xref ref-type="table" rid="T6">Table 6</xref>, demonstrates that the MUNet model performs well across several key performance metrics. The model exhibits high accuracy and balanced accuracy, along with excellent performance in specificity and AUC, indicating its robustness and generalization ability in the LGG segmentation task. These results validate the effectiveness of MUNet in handling medical image segmentation tasks. <xref ref-type="fig" rid="F5">Figure 5</xref> visualizes the LGG segmentation dataset, and from the figure, the segmentation results of the model can be intuitively observed. MUNet is able to accurately identify brain tumor segmentation regions, and the overlap between the segmentation results and the ground truth is highly consistent, further demonstrating the model&#x00027;s superior performance and reliability. This high-quality segmentation result also lays a solid foundation for automated processing in medical image analysis.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>MUNet independent validation on the LGG segmentation dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>Kappa</bold></th>
<th valign="top" align="center"><bold>DSC</bold></th>
<th valign="top" align="center"><bold>IOU</bold></th>
<th valign="top" align="center"><bold>Sensitivity</bold></th>
<th valign="top" align="center"><bold>Specificity</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Accuracy</bold></th>
<th valign="top" align="center"><bold>BA</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>MUNet</bold></td>
<td valign="top" align="center">0.678</td>
<td valign="top" align="center">0.702</td>
<td valign="top" align="center">0.581</td>
<td valign="top" align="center">0.658</td>
<td valign="top" align="center">0.975</td>
<td valign="top" align="center">0.762</td>
<td valign="top" align="center">0.981</td>
<td valign="top" align="center">0.808</td>
<td valign="top" align="center">0.812</td>
</tr></tbody>
</table>
</table-wrap>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Display of MUNet segmentation performance on the LGG segmentation dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fncom-19-1513059-g0005.tif"/>
</fig>
</sec>
<sec>
<title>4.5 Limitations and future work</title>
<p>Although MUNet performs well in the task of brain tumor segmentation, there are still some limitations that require further research and improvement. Firstly, while the SD-SSM block in MUNet effectively combines global and local features, the model&#x00027;s accuracy may decline when dealing with extremely complex or highly heterogeneous tumor regions. In such cases, unclear tumor boundaries or irregular shapes may result in insufficient precision in segmentation. Future work could focus on enhancing boundary detection mechanisms to improve boundary recognition in complex tumor regions.</p>
<p>Secondly, as a black-box model, MUNet lacks interpretability, which may limit its widespread application in clinical settings. Although MUNet performs excellently in brain tumor segmentation tasks, in clinical practice, doctors often need to understand the decision-making process of the model, especially when faced with critical diagnoses (Dasanayaka et al., <xref ref-type="bibr" rid="B13">2022a</xref>). Future research could focus on designing transparent reasoning processes or explanation modules within the model, allowing doctors to clearly understand each decision step and the weight distribution, thereby increasing trust in the model&#x00027;s predictions.</p>
<p>In addition, although MUNet has achieved good performance on the current dataset, it still faces the issue of overfitting, especially when the data is limited or the data distribution is imbalanced. Overfitting may result in good performance on the training set, but poor predictive performance on unseen data in practical applications. To reduce the risk of overfitting, future research can enhance the diversity of the dataset by increasing the amount of annotated data or employing data augmentation techniques to expand the training set. Meanwhile, regularization methods can help reduce model complexity and decrease the probability of overfitting. Additionally, transfer learning could be employed, where the model is pre-trained on large publicly available datasets and then fine-tuned for the specific brain tumor segmentation task, thus improving the model&#x00027;s generalization ability.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>This paper presents a novel network framework named MUNet, which combines the advantages of UNet and Mamba to achieve efficient and accurate brain tumor segmentation. By introducing the SD-SSM module, which utilizes selective scanning and state space modeling, MUNet can capture both global and local features of images, thereby improving segmentation accuracy. Additionally, the integration of the SD-Conv structure reduces feature redundancy without increasing the number of parameters, enhancing the overall efficiency of the model. Experimental results show that MUNet outperforms existing methods on the BraTS2020 and BraTS2018 datasets, achieving superior segmentation accuracy. Moreover, MUNet demonstrates excellent generalization capabilities when validated on the independent LGG segmentation dataset, further proving its effectiveness in various medical imaging scenarios. Future work may focus on extending the application of MUNet to other imaging modalities and exploring more advanced learning strategies to further enhance its clinical applicability.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: BraTS2020 Dataset: <ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/awsaf49/brats20-dataset-training-validation.BraTS2018">https://www.kaggle.com/datasets/awsaf49/brats20-dataset-training-validation.BraTS2018</ext-link>, Dataset: <ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/anassbenfares/brats2018.LGG_Segmentation">https://www.kaggle.com/datasets/anassbenfares/brats2018.LGGSegmentation</ext-link>, and Dataset: <ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/mateuszbuda/lgg-mri-segmentation">https://www.kaggle.com/datasets/mateuszbuda/lgg-mri-segmentation</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>LY: Conceptualization, Methodology, Project administration, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing, Formal analysis. QD: Conceptualization, Formal analysis, Methodology, Project administration, Visualization, Writing &#x02013; review &#x00026; editing. DL: Data curation, Formal analysis, Investigation, Methodology, Project administration, Validation, Writing &#x02013; review &#x00026; editing. CT: Data curation, Investigation, Project administration, Resources, Visualization, Writing &#x02013; original draft. XL: Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Validation, Writing &#x02013; review &#x00026; editing, Visualization, Writing &#x02013; original draft.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by the National Traditional Chinese Medicine Key Specialty Construction Project (Document No. [2024] 90, issued by the Medical Administration and Supervision Department of the National Administration of Traditional Chinese Medicine).</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abhisheka</surname> <given-names>B.</given-names></name> <name><surname>Biswas</surname> <given-names>S. K.</given-names></name> <name><surname>Purkayastha</surname> <given-names>B.</given-names></name></person-group> (<year>2023</year>). <article-title>A comprehensive review on breast cancer detection, classification and segmentation using deep learning</article-title>. <source>Arch. Comput. Methods Eng</source>. <volume>30</volume>, <fpage>5023</fpage>&#x02013;<lpage>5052</lpage>. <pub-id pub-id-type="doi">10.1007/s11831-023-09968-z</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Almotairi</surname> <given-names>S.</given-names></name> <name><surname>Kareem</surname> <given-names>G.</given-names></name> <name><surname>Aouf</surname> <given-names>M.</given-names></name> <name><surname>Almutairi</surname> <given-names>B.</given-names></name> <name><surname>Salem</surname> <given-names>M. A.-M.</given-names></name></person-group> (<year>2020</year>). <article-title>Liver tumor segmentation in CT scans using modified SegNet</article-title>. <source>Sensors</source> <volume>20</volume>:<fpage>1516</fpage>. <pub-id pub-id-type="doi">10.3390/s20051516</pub-id><pub-id pub-id-type="pmid">32164153</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Amgad</surname> <given-names>M.</given-names></name> <name><surname>Atteya</surname> <given-names>L. A.</given-names></name> <name><surname>Hussein</surname> <given-names>H.</given-names></name> <name><surname>Mohammed</surname> <given-names>K. H.</given-names></name> <name><surname>Hafiz</surname> <given-names>E.</given-names></name> <name><surname>Elsebaie</surname> <given-names>M. A.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>NUCLS: a scalable crowdsourcing approach and dataset for nucleus classification and segmentation in breast cancer</article-title>. <source>GigaScience</source> <volume>11</volume>:<fpage>giac037</fpage>. <pub-id pub-id-type="doi">10.1093/gigascience/giac037</pub-id><pub-id pub-id-type="pmid">35579553</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Badiezadeh</surname> <given-names>A.</given-names></name> <name><surname>Malekmohammadi</surname> <given-names>A.</given-names></name> <name><surname>Mirhassani</surname> <given-names>S. M.</given-names></name> <name><surname>Gifani</surname> <given-names>P.</given-names></name> <name><surname>Vafaeezadeh</surname> <given-names>M.</given-names></name></person-group> (<year>2024</year>). <article-title>Segmentation strategies in deep learning for prostate cancer diagnosis: a comparative study of mamba, sam, and yolo</article-title>. <source>arXiv preprint arXiv:2409.16205</source>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bakas</surname> <given-names>S.</given-names></name> <name><surname>Akbari</surname> <given-names>H.</given-names></name> <name><surname>Sotiras</surname> <given-names>A.</given-names></name> <name><surname>Bilello</surname> <given-names>M.</given-names></name> <name><surname>Rozycki</surname> <given-names>M.</given-names></name> <name><surname>Kirby</surname> <given-names>J. S.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Advancing the cancer genome atlas glioma MRI collections with expert segmentation labels and radiomic features</article-title>. <source>Sci. Data</source> <volume>4</volume>, <fpage>1</fpage>&#x02013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1038/sdata.2017.117</pub-id><pub-id pub-id-type="pmid">28872634</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bakas</surname> <given-names>S.</given-names></name> <name><surname>Reyes</surname> <given-names>M.</given-names></name> <name><surname>Jakab</surname> <given-names>A.</given-names></name> <name><surname>Bauer</surname> <given-names>S.</given-names></name> <name><surname>Rempfler</surname> <given-names>M.</given-names></name> <name><surname>Crimi</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Identifying the best machine learning algorithms for brain tumor segmentation, progression assessment, and overall survival prediction in the brats challenge</article-title>. <source>arXiv preprint arXiv:1811.02629</source>.</citation>
</ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Belaid</surname> <given-names>O. N.</given-names></name> <name><surname>Loudini</surname> <given-names>M.</given-names></name> <name><surname>Nakib</surname> <given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;Brain tumor classification using denseNet and u-net convolutional neural networks,&#x0201D;</article-title> in <source>2024 8th International Conference on Image and Signal Processing and their Applications (ISPA)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1109/ISPA59904.2024.10536704</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Buda</surname> <given-names>M.</given-names></name> <name><surname>Saha</surname> <given-names>A.</given-names></name> <name><surname>Mazurowski</surname> <given-names>M. A.</given-names></name></person-group> (<year>2019</year>). <source>LGG Segmentation Dataset</source>. <publisher-loc>San Francisco</publisher-loc>: <publisher-name>Kaggle</publisher-name>.</citation>
</ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cao</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Jiang</surname> <given-names>D.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Tian</surname> <given-names>Q.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;Swin-unet: Unet-like pure transformer for medical image segmentation,&#x0201D;</article-title> in <source>European Conference on Computer Vision</source> (<publisher-loc>Springer</publisher-loc>), <fpage>205</fpage>&#x02013;<lpage>218</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-25066-8_9</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chaudhury</surname> <given-names>S.</given-names></name> <name><surname>Krishna</surname> <given-names>A. N.</given-names></name> <name><surname>Gupta</surname> <given-names>S.</given-names></name> <name><surname>Sankaran</surname> <given-names>K. S.</given-names></name> <name><surname>Khan</surname> <given-names>S.</given-names></name> <name><surname>Sau</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>[retracted] effective image processing and segmentation-based machine learning techniques for diagnosis of breast cancer</article-title>. <source>Comput. Math. Methods Med</source>. <volume>2022</volume>:<fpage>6841334</fpage>. <pub-id pub-id-type="doi">10.1155/2022/6841334</pub-id><pub-id pub-id-type="pmid">35432588</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Lu</surname> <given-names>Y.</given-names></name> <name><surname>Yu</surname> <given-names>Q.</given-names></name> <name><surname>Luo</surname> <given-names>X.</given-names></name> <name><surname>Adeli</surname> <given-names>E.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>TransUNet: transformers make strong encoders for medical image segmentation</article-title>. <source>arXiv preprint arXiv:2102.04306</source>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chinnam</surname> <given-names>S. K. R.</given-names></name> <name><surname>Sistla</surname> <given-names>V.</given-names></name> <name><surname>Kolli</surname> <given-names>V. K. K.</given-names></name></person-group> (<year>2022</year>). <article-title>Multimodal attention-gated cascaded U-Net model for automatic brain tumor detection and segmentation</article-title>. <source>Biomed. Signal Process. Control</source> <volume>78</volume>:<fpage>103907</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2022.103907</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dasanayaka</surname> <given-names>S.</given-names></name> <name><surname>Shantha</surname> <given-names>V.</given-names></name> <name><surname>Silva</surname> <given-names>S.</given-names></name> <name><surname>Meedeniya</surname> <given-names>D.</given-names></name> <name><surname>Ambegoda</surname> <given-names>T.</given-names></name></person-group> (<year>2022a</year>). <article-title>Interpretable machine learning for brain tumour analysis using MRI and whole slide images</article-title>. <source>Softw. Impacts</source> <volume>13</volume>:<fpage>100340</fpage>. <pub-id pub-id-type="doi">10.1016/j.simpa.2022.100340</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dasanayaka</surname> <given-names>S.</given-names></name> <name><surname>Silva</surname> <given-names>S.</given-names></name> <name><surname>Shantha</surname> <given-names>V.</given-names></name> <name><surname>Meedeniya</surname> <given-names>D.</given-names></name> <name><surname>Ambegoda</surname> <given-names>T.</given-names></name></person-group> (<year>2022b</year>). <article-title>&#x0201C;Interpretable machine learning for brain tumor analysis using MRI,&#x0201D;</article-title> in <source>2022 2nd International Conference on Advanced Research in Computing (ICARC)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>212</fpage>&#x02013;<lpage>217</lpage>. <pub-id pub-id-type="doi">10.1109/ICARC54489.2022.9754131</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Futrega</surname> <given-names>M.</given-names></name> <name><surname>Milesi</surname> <given-names>A.</given-names></name> <name><surname>Marcinkiewicz</surname> <given-names>M.</given-names></name> <name><surname>Ribalta</surname> <given-names>P.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Optimized U-Net for brain tumor segmentation,&#x0201D;</article-title> in <source>International MICCAI Brainlesion Workshop</source> (<publisher-loc>Springer</publisher-loc>), <fpage>15</fpage>&#x02013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-09002-8_2</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ho</surname> <given-names>D. J.</given-names></name> <name><surname>Yarlagadda</surname> <given-names>D. V.</given-names></name> <name><surname>D&#x00027;Alfonso</surname> <given-names>T. M.</given-names></name> <name><surname>Hanna</surname> <given-names>M. G.</given-names></name> <name><surname>Grabenstetter</surname> <given-names>A.</given-names></name> <name><surname>Ntiamoah</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Deep multi-magnification networks for multi-class breast cancer image segmentation</article-title>. <source>Comput. Med. Imag. Graph</source>. <volume>88</volume>:<fpage>101866</fpage>. <pub-id pub-id-type="doi">10.1016/j.compmedimag.2021.101866</pub-id><pub-id pub-id-type="pmid">33485058</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Houssein</surname> <given-names>E. H.</given-names></name> <name><surname>Emam</surname> <given-names>M. M.</given-names></name> <name><surname>Ali</surname> <given-names>A. A.</given-names></name></person-group> (<year>2021</year>). <article-title>An efficient multilevel thresholding segmentation method for thermography breast cancer imaging based on improved chimp optimization algorithm</article-title>. <source>Expert Syst. Appl</source>. <volume>185</volume>:<fpage>115651</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2021.115651</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>T.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Jiang</surname> <given-names>L.</given-names></name></person-group> (<year>2023</year>). <article-title>Ds-UNext: depthwise separable convolution network with large convolutional kernel for medical image segmentation</article-title>. <source>Signal, Image Video Proc</source>. <volume>17</volume>, <fpage>1775</fpage>&#x02013;<lpage>1783</lpage>. <pub-id pub-id-type="doi">10.1007/s11760-022-02388-9</pub-id><pub-id pub-id-type="pmid">39503115</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ibrahim</surname> <given-names>A.</given-names></name> <name><surname>Mohammed</surname> <given-names>S.</given-names></name> <name><surname>Ali</surname> <given-names>H. A.</given-names></name> <name><surname>Hussein</surname> <given-names>S. E.</given-names></name></person-group> (<year>2020</year>). <article-title>Breast cancer segmentation from thermal images based on chaotic salp swarm algorithm</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>122121</fpage>&#x02013;<lpage>122134</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.3007336</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jyothi</surname> <given-names>P.</given-names></name> <name><surname>Singh</surname> <given-names>A. R.</given-names></name></person-group> (<year>2023</year>). <article-title>Deep learning models and traditional automated techniques for brain tumor segmentation in MRI: a review</article-title>. <source>Artif. Intell. Rev</source>. <volume>56</volume>, <fpage>2923</fpage>&#x02013;<lpage>2969</lpage>. <pub-id pub-id-type="doi">10.1007/s10462-022-10245-x</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>Y.</given-names></name> <name><surname>Kim</surname> <given-names>C.</given-names></name></person-group> (<year>2024</year>). <article-title>Wave-u-mamba: an end-to-end framework for high-quality and efficient speech super resolution</article-title>. <source>arXiv preprint arXiv:2409.09337</source>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Wen</surname> <given-names>Y.</given-names></name> <name><surname>He</surname> <given-names>L.</given-names></name></person-group> (<year>2023a</year>). <article-title>&#x0201C;Scconv: spatial and channel reconstruction convolution for feature redundancy,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 6153&#x02013;6162. <pub-id pub-id-type="doi">10.1109/CVPR52729.2023.00596</pub-id><pub-id pub-id-type="pmid">38379942</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>P.</given-names></name> <name><surname>Guo</surname> <given-names>D.</given-names></name> <name><surname>Lu</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2023b</year>). <article-title>LViT: language meets vision transformer in medical image segmentation</article-title>. <source>IEEE Trans. Med. Imag</source>. <volume>43</volume>, <fpage>96</fpage>&#x02013;<lpage>107</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2023.3291719</pub-id><pub-id pub-id-type="pmid">37399157</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Magadza</surname> <given-names>T.</given-names></name> <name><surname>Viriri</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep learning for brain tumor segmentation: a survey of state-of-the-art</article-title>. <source>J. Imag</source>. <volume>7</volume>:<fpage>19</fpage>. <pub-id pub-id-type="doi">10.3390/jimaging7020019</pub-id><pub-id pub-id-type="pmid">34460618</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maji</surname> <given-names>D.</given-names></name> <name><surname>Sigedar</surname> <given-names>P.</given-names></name> <name><surname>Singh</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>Attention res-UNet with guided decoder for semantic segmentation of brain tumors</article-title>. <source>Biomed. Signal Process. Control</source> <volume>71</volume>:<fpage>103077</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2021.103077</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Menze</surname> <given-names>B. H.</given-names></name> <name><surname>Jakab</surname> <given-names>A.</given-names></name> <name><surname>Bauer</surname> <given-names>S.</given-names></name> <name><surname>Kalpathy-Cramer</surname> <given-names>J.</given-names></name> <name><surname>Farahani</surname> <given-names>K.</given-names></name> <name><surname>Kirby</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>The multimodal brain tumor image segmentation benchmark (brats)</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>34</volume>, <fpage>1993</fpage>&#x02013;<lpage>2024</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2014.2377694</pub-id><pub-id pub-id-type="pmid">25494501</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Michael</surname> <given-names>E.</given-names></name> <name><surname>Ma</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Kulwa</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Breast cancer segmentation methods: current status and future potentials</article-title>. <source>Biomed. Res. Int</source>. <volume>2021</volume>:<fpage>9962109</fpage>. <pub-id pub-id-type="doi">10.1155/2021/9962109</pub-id><pub-id pub-id-type="pmid">34337066</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nizamani</surname> <given-names>A. H.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Nizamani</surname> <given-names>A. A.</given-names></name> <name><surname>Bhatti</surname> <given-names>U. A.</given-names></name></person-group> (<year>2023</year>). <article-title>Advance brain tumor segmentation using feature fusion methods with deep u-net model with CNN for MRI data</article-title>. <source>J. King Saud Univ. Comput. Inf. Sci</source>. <volume>35</volume>:<fpage>101793</fpage>. <pub-id pub-id-type="doi">10.1016/j.jksuci.2023.101793</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oktay</surname> <given-names>O.</given-names></name> <name><surname>Schlemper</surname> <given-names>J.</given-names></name> <name><surname>Folgoc</surname> <given-names>L. L.</given-names></name> <name><surname>Lee</surname> <given-names>M.</given-names></name> <name><surname>Heinrich</surname> <given-names>M.</given-names></name> <name><surname>Misawa</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Attention U-Net: learning where to look for the pancreas</article-title>. <source>arXiv preprint arXiv:1804.03999</source>.<pub-id pub-id-type="pmid">35474556</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Patro</surname> <given-names>B. N.</given-names></name> <name><surname>Agneeswaran</surname> <given-names>V. S.</given-names></name></person-group> (<year>2024</year>). <article-title>Simba: simplified mamba-based architecture for vision and multivariate time series</article-title>. <source>arXiv preprint arXiv:2403.15360</source>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ranjbarzadeh</surname> <given-names>R.</given-names></name> <name><surname>Bagherian Kasgari</surname> <given-names>A.</given-names></name> <name><surname>Jafarzadeh Ghoushchi</surname> <given-names>S.</given-names></name> <name><surname>Anari</surname> <given-names>S.</given-names></name> <name><surname>Naseri</surname> <given-names>M.</given-names></name> <name><surname>Bendechache</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>Brain tumor segmentation based on deep learning and an attention mechanism using MRI multi-modalities brain images</article-title>. <source>Sci. Rep</source>. <volume>11</volume>, <fpage>1</fpage>&#x02013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-021-90428-8</pub-id><pub-id pub-id-type="pmid">34035406</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rezaei</surname> <given-names>Z.</given-names></name></person-group> (<year>2021</year>). <article-title>A review on image-based approaches for breast cancer detection, segmentation, and classification</article-title>. <source>Expert Syst. Appl</source>. <volume>182</volume>:<fpage>115204</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2021.115204</pub-id><pub-id pub-id-type="pmid">26171249</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Soulami</surname> <given-names>K. B.</given-names></name> <name><surname>Kaabouch</surname> <given-names>N.</given-names></name> <name><surname>Saidi</surname> <given-names>M. N.</given-names></name> <name><surname>Tamtaoui</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>Breast cancer: one-stage automated detection, segmentation, and classification of digital mammograms using UNet model based-semantic segmentation</article-title>. <source>Biomed. Signal Process. Control</source> <volume>66</volume>:<fpage>102481</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2021.102481</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>H.</given-names></name> <name><surname>Cheng</surname> <given-names>L.</given-names></name> <name><surname>Huang</surname> <given-names>G.</given-names></name> <name><surname>Tan</surname> <given-names>Z.</given-names></name> <name><surname>Lu</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>Rotate to scan: UNet-like mamba with triplet SSM module for medical image segmentation</article-title>. <source>arXiv preprint arXiv:2403.17701</source>. <pub-id pub-id-type="doi">10.1007/s11760-024-03484-8</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Walsh</surname> <given-names>J.</given-names></name> <name><surname>Othmani</surname> <given-names>A.</given-names></name> <name><surname>Jain</surname> <given-names>M.</given-names></name> <name><surname>Dev</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>Using u-net network for efficient brain tumor segmentation in MRI images</article-title>. <source>Healthcare Anal</source>. <volume>2</volume>:<fpage>100098</fpage>. <pub-id pub-id-type="doi">10.1016/j.health.2022.100098</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Zheng</surname> <given-names>Y.</given-names></name> <name><surname>Ma</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Gee</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Information bottleneck-based interpretable multitask network for breast cancer classification and segmentation</article-title>. <source>Med. Image Anal</source>. <volume>83</volume>:<fpage>102687</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2022.102687</pub-id><pub-id pub-id-type="pmid">36436356</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Zheng</surname> <given-names>J.-Q.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Cui</surname> <given-names>G.</given-names></name> <name><surname>Li</surname> <given-names>L.</given-names></name></person-group> (<year>2024</year>). <article-title>Mamba-UNet: UNet-like pure visual mamba for medical image segmentation</article-title>. <source>arXiv preprint arXiv:2402.05079</source>.</citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wijethilake</surname> <given-names>N.</given-names></name> <name><surname>Meedeniya</surname> <given-names>D.</given-names></name> <name><surname>Chitraranjan</surname> <given-names>C.</given-names></name> <name><surname>Perera</surname> <given-names>I.</given-names></name> <name><surname>Islam</surname> <given-names>M.</given-names></name> <name><surname>Ren</surname> <given-names>H.</given-names></name></person-group> (<year>2021</year>). <article-title>Glioma survival analysis empowered with data engineering-a survey</article-title>. <source>IEEE Access</source> <volume>9</volume>, <fpage>43168</fpage>&#x02013;<lpage>43191</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2021.3065965</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>J.</given-names></name></person-group> (<year>2024</year>). <article-title>HC-mamba: vision mamba with hybrid convolutional techniques for medical image segmentation</article-title>. <source>arXiv preprint arXiv:2405.05007</source>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>R.</given-names></name> <name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Du</surname> <given-names>B.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name></person-group> (<year>2024</year>). <article-title>A survey on vision mamba: models, applications and challenges</article-title>. <source>arXiv preprint arXiv:2404.18861</source>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zebari</surname> <given-names>D. A.</given-names></name> <name><surname>Zeebaree</surname> <given-names>D. Q.</given-names></name> <name><surname>Abdulazeez</surname> <given-names>A. M.</given-names></name> <name><surname>Haron</surname> <given-names>H.</given-names></name> <name><surname>Hamed</surname> <given-names>H. N. A.</given-names></name></person-group> (<year>2020</year>). <article-title>Improved threshold based and trainable fully automated segmentation for breast cancer boundary and pectoral muscle in mammogram images</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>203097</fpage>&#x02013;<lpage>203116</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.3036072</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Gu</surname> <given-names>L.</given-names></name> <name><surname>Ling</surname> <given-names>T.</given-names></name> <name><surname>Tao</surname> <given-names>X.</given-names></name></person-group> (<year>2024</year>). <article-title>HMT-UNet: a hybird mamba-transformer vision UNet for medical image segmentation</article-title>. <source>arXiv preprint arXiv:2408.11289</source>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>L.</given-names></name> <name><surname>Liao</surname> <given-names>B.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name></person-group> (<year>2024</year>). <article-title>Vision mamba: efficient visual representation learning with bidirectional state space model</article-title>. <source>arXiv preprint arXiv:2401.09417</source>.</citation>
</ref>
</ref-list>
</back>
</article>