<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Phys.</journal-id>
<journal-title>Frontiers in Physics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Phys.</abbrev-journal-title>
<issn pub-type="epub">2296-424X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1532638</article-id>
<article-id pub-id-type="doi">10.3389/fphy.2024.1532638</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Multi-Conv attention network for skin lesion image segmentation</article-title>
<alt-title alt-title-type="left-running-head">Li et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphy.2024.1532638">10.3389/fphy.2024.1532638</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Zexin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2843677/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Hanchen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Haoyu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2657812/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lin</surname>
<given-names>Chenxin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2912370/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yan</surname>
<given-names>Aochen</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2902075/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>International College</institution>, <institution>Chongqing University of Posts and Telecommunications</institution>, <addr-line>Chongqing</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Viterbi School of Engineering</institution>, <institution>University of Southern California</institution>, <addr-line>Los Angeles</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/669638/overview">Bo Xiao</ext-link>, Imperial College London, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2059143/overview">Gang Hu</ext-link>, Buffalo State College, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2661383/overview">Yafei Zhang</ext-link>, Kunming University of Science and Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2882357/overview">Yimin Chen</ext-link>, University of Massachusetts Lowell, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Aochen Yan, <email>aochenya@usc.edu</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>20</day>
<month>12</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1532638</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>12</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Li, Wang, Chen, Lin and Yan.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Li, Wang, Chen, Lin and Yan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>To address the trade-off between segmentation performance and model lightweighting in computer-aided skin lesion segmentation, this paper proposes a lightweight network architecture, Multi-Conv Attention Network (MCAN). The network consists of two key modules: ISDConv (Inception-Split Depth Convolution) and AEAM (Adaptive Enhanced Attention Module). ISDConv reduces computational complexity by decomposing large kernel depthwise convolutions into smaller kernel convolutions and unit mappings. The AEAM module leverages dimensional decoupling, lightweight multi-semantic guidance, and semantic discrepancy alleviation to facilitate the synergy between channel attention and spatial attention, further exploiting redundancy in the spatial and channel feature maps. With these improvements, the proposed method achieves a balance between segmentation performance and computational efficiency. Experimental results demonstrate that MCAN achieves state-of-the-art performance on mainstream skin lesion segmentation datasets, validating its effectiveness.</p>
</abstract>
<kwd-group>
<kwd>medical image segmentation</kwd>
<kwd>lightweight</kwd>
<kwd>melanoma</kwd>
<kwd>attention mechanism</kwd>
<kwd>Inception</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Radiation Detectors and Imaging</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Melanoma, a highly malignant skin tumor, causes a significant number of deaths worldwide each year. Its incidence and mortality rates vary significantly depending on the region, the level of early diagnosis awareness, and the accessibility of primary care [<xref ref-type="bibr" rid="B1">1</xref>]. Early detection of melanoma is crucial for improving patient survival rates. However, due to the diversity and complexity of melanoma&#x2019;s appearance, its accurate diagnosis often relies on the experience and expertise of doctors, which somewhat limits the efficiency and accuracy of early diagnosis.</p>
<p>In melanoma diagnosis, image segmentation is a key step that precisely separates the lesion area from healthy skin, helping doctors identify the lesion&#x2019;s boundaries and assist in accurate diagnosis and treatment. Traditional segmentation methods rely heavily on complex preprocessing and manual feature extraction, making it difficult to handle the complexity of melanoma images. With the emergence of high-quality datasets, data-driven deep learning methods have rapidly gained popularity. Zhang et al. [<xref ref-type="bibr" rid="B2">2</xref>] proposed a novel framework that integrates multiple experts to jointly learn representations from diverse MRI modalities, effectively enhancing segmentation performance. Similarly, Li et al. [<xref ref-type="bibr" rid="B3">3</xref>] addressed challenges in brain tumor segmentation caused by missing modalities by utilizing a deformation-aware learning framework that reconstructs missing information, resulting in more reliable and accurate segmentation even in incomplete datasets. Among them, attention mechanisms, as an effective way to integrate local and global features, help the model focus on the lesion areas. Dong et al. [<xref ref-type="bibr" rid="B4">4</xref>] enhanced the capability to capture feature information by dynamically allocating attention weights across channel and spatial dimensions, addressing the complex features, blurry boundaries, and noise interference in skin lesion segmentation. Similarly, the GL-CSAM module designed by Sun et al. [<xref ref-type="bibr" rid="B5">5</xref>] aims to capture global contextual information, enhancing the model&#x2019;s ability to perceive global features. However, they did not fully explore feature fusion between different convolutional layers. To address this issue, Qiu et al. [<xref ref-type="bibr" rid="B6">6</xref>] introduced a multi-level attention fusion mechanism that progressively extracts lesion boundary information using contextual information from different levels, alleviating the problem of blurry boundaries. Qi et al. [<xref ref-type="bibr" rid="B7">7</xref>] and Liu et al. [<xref ref-type="bibr" rid="B8">8</xref>] introduced single attention mechanisms to integrate contextual features, specifically designed for stroke lesion segmentation. The combination of standalone self-attention modules with convolutional layers has shown limited effectiveness in enhancing the model&#x2019;s non-local feature modeling capabilities. To address this limitation, Yang et al. [<xref ref-type="bibr" rid="B9">9</xref>] introduced a multi-attention mechanism (spatial and reverse attention). Spatial attention is used to improve the extraction of useful features, while reverse attention enhances the network&#x2019;s segmentation performance by applying reverse attention operations on skip connections, enabling more accurate analysis and localization of small lesion targets. Liu et al. [<xref ref-type="bibr" rid="B10">10</xref>] and Zhu et al. [<xref ref-type="bibr" rid="B11">11</xref>] enhanced the precision and detail of tumor segmentation by fusing information from multiple MRI modes such as T1, T2, and FLAIR. Zhu et al. [<xref ref-type="bibr" rid="B12">12</xref>] embedded a feature fusion module based on attention mechanism in the model structure to optimize the expression and integration of multi-modal features to improve segmentation accuracy. Liu et al. [<xref ref-type="bibr" rid="B13">13</xref>] examined the effectiveness of traditional objective evaluation indicators in the evaluation of image fusion results and proposed a statiscy-based framework to compensate for the shortcomings of existing indicators. These methods have improved the segmentation task to varying degrees at different stages, achieving commendable results. However, their network designs do not fully consider how to effectively utilize spatial information, and they lack dedicated mechanisms to enhance and preserve spatial information. These shortcomings may result in suboptimal performance when handling spatial correlations.</p>
<p>Moreover, it is worth noting that while introducing high-quality attention mechanisms, the parameter count of the model increases, potentially compromising the real-time performance during deployment. Although high-quality attention mechanisms can enhance model performance, they are often accompanied by an increase in parameter count, which can negatively impact the real-time performance of model deployment [<xref ref-type="bibr" rid="B14">14</xref>]. In response to such problems, most researchers have based their efforts on the potential of deep separable convolution to improve model efficiency and effectiveness. Zhou et al. [<xref ref-type="bibr" rid="B15">15</xref>] constructs expansion layers using depthwise separable convolutions to efficiently extract multi-scale features with low computational overhead, enhancing the feature representation capability. Liu et al. [<xref ref-type="bibr" rid="B16">16</xref>], Ma et al. [<xref ref-type="bibr" rid="B17">17</xref>], and Feng et al. [<xref ref-type="bibr" rid="B18">18</xref>] adopted a similar approach by integrating depthwise separable convolution layers into the encoder. However, they often struggle to achieve precise detailed description while maintaining low computational overhead. Ruan et al. [<xref ref-type="bibr" rid="B19">19</xref>] combined MLP to extract global feature information, followed by feature extraction using depthwise separable convolutions (DWConv). This effectively preserved significant features in the brain feature map while filtering out less relevant features. However, the lightweight processing of complex features remains limited. Similarly, Lei et al. [<xref ref-type="bibr" rid="B20">20</xref>] combined depthwise separable convolutions with bilinear interpolation to adjust the size of high-level features, making them match low-level features. However, this approach faces performance bottlenecks when further reducing the computational burden. Chen et al. [<xref ref-type="bibr" rid="B21">21</xref>] incorporated the advantages of asymmetric convolutions based on depthwise separable convolutions and designed an ultralight convolution module, further achieving the decoupling of spatial and channel dimensions. Existing methods still have limitations in lightweight design. Although different encoder designs effectively reduce computational load and ensure efficient feature extraction, they still lack precision in representing the blurry edges of skin lesions.</p>
<p>To address the contradiction between segmentation performance and lightweight design, this paper proposes a lightweight segmentation method. It aims to more accurately capture and segment the lesion area by leveraging channel and spatial redundancy, without increasing additional computational load. Specifically, the core of the segmentation framework is the Inception-Split ISDConv. Additionally, at the bridging layer stage, we introduce the AEAM, which combines the collaborative effects of spatial and channel attention with the feature calibration capabilities of the squeeze-and-excitation network. AEAM utilizes multi-scale depth-shared 1D convolutions to capture multi-semantic spatial information for each feature channel. It effectively integrates global contextual dependencies and multi-semantic spaces, while calculating channel similarity and contributions under the guidance of compressed spatial knowledge, thereby alleviating semantic differences in the spatial structure. Additionally, we introduce dynamic convolution in the encoder. Dynamic convolution dynamically aggregates multiple parallel convolution kernels based on input-relevant attention mechanisms. Assembling multiple convolution kernels is not only computationally efficient but also enhances representational capability due to the smaller size of the kernels.</p>
<p>The contributions of this paper can be summarized in the following three aspects:<list list-type="simple">
<list-item>
<p>1. In this study, a novel lightweight segmentation network named Multi-Conv Attention Network (MCAN) is proposed. It performs channel and spatial weighting on the spatial and channel redundancies in the feature map without increasing additional computational load, achieving an effect of information complementarity.</p>
</list-item>
<list-item>
<p>2. To address the unclear edges in skin lesions, this paper proposes ISDConv. This module performs multi-scale feature extraction using depthwise separable convolutions, multi-scale convolution kernels, and spatial and channel reconstruction convolutions. It reduces computational complexity and the number of parameters, thereby improving the model&#x2019;s feature representation capability while maintaining efficient feature extraction.</p>
</list-item>
<list-item>
<p>3. To address the insufficient utilization of redundancies in the spatial and channel feature maps, this paper proposes the Adaptive Enhanced Attention Module (AEAM). Through dimension decoupling, lightweight multi-semantic guidance, and semantic discrepancy mitigation, AEAM achieves the collaborative effect between channel and spatial attention, enabling the model to capture and segment the lesion areas more accurately.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2">
<title>Related works</title>
<sec id="s2-1">
<title>Attention mechanism</title>
<p>In the field of natural images, Li et al. [<xref ref-type="bibr" rid="B22">22</xref>] used a dual attention fusion module to effectively combine features from images from different sources, thereby enhancing the model&#x2019;s ability to focus on important regions. The attention mechanism can enhance the extraction of key features in infrared and visible images, making the fused images clearer and retaining more meaningful details [<xref ref-type="bibr" rid="B23">23</xref>]. In medical image segmentation, the attention mechanism is primarily used to guide the model&#x2019;s focus on the lesion areas in the image, assigning different weights to each pixel or feature, enhancing task-relevant features, and suppressing irrelevant background information. Huang et al. [<xref ref-type="bibr" rid="B24">24</xref>] prior convolutional attention mechanism that dynamically allocates attention weights across both channel and spatial dimensions. Shaker et al. [<xref ref-type="bibr" rid="B25">25</xref>] used a pair of mutually dependent branches based on spatial and channel attention to effectively learn discriminative features, improving the quality of segmentation masks. Fu et al. [<xref ref-type="bibr" rid="B26">26</xref>] used a Transformer-based spatial and channel attention module to extract global complementary information across different layers of the U-Net, which helps in learning detailed features at different scales. To address hair interference in dermoscopic images, Xiong et al. [<xref ref-type="bibr" rid="B27">27</xref>] proposed a multi-scale channel attention mechanism that enhances feature information and boundary awareness. Song et al. [<xref ref-type="bibr" rid="B28">28</xref>] argued that current popular attention mechanisms focus too much on external image features and lack research on latent features. They introduced an external-latent attention mechanism, using an entropy quantization method to summarize the distribution of latent contextual information. Similarly, Huang et al. [<xref ref-type="bibr" rid="B29">29</xref>] used Bi-Level Routing Attention in deep networks to discard irrelevant key-value pairs, achieving content-aware sparse attention for dispersed semantic information.</p>
</sec>
<sec id="s2-2">
<title>Network lightweighting</title>
<p>While pursuing high performance, researchers have also begun to focus on the lightweight and efficiency of medical image segmentation networks. Network structure design is one of the most popular approaches for lightweight optimization. Ma et al. [<xref ref-type="bibr" rid="B17">17</xref>] simplified the structure, reduced the number of parameters, and optimized the convolution operations, achieving a significant reduction in computational complexity and model size while maintaining segmentation accuracy. This enables the model to perform excellently even in resource-constrained environments, making it suitable for applications such as mobile healthcare and telemedicine. The UcUNet [<xref ref-type="bibr" rid="B30">30</xref>] network achieves lightweight and precise medical image segmentation by designing an efficient large-kernel U-shaped convolution module. This network leverages large-kernel convolutions to expand the receptive field while integrating depthwise separable convolutions to reduce the computational cost, thereby maintaining high segmentation accuracy with efficient computation. Liu et al. [<xref ref-type="bibr" rid="B16">16</xref>] combines the lightweight characteristics of HarDNet with multi-attention mechanisms, enhancing the network&#x2019;s ability to capture key features and achieving more precise medical image segmentation. Sun et al. [<xref ref-type="bibr" rid="B31">31</xref>] introduces a contextual residual network, effectively integrating contextual information into the U-shaped network, enhancing the global understanding and stability of the segmentation. Nisa and Ismail [<xref ref-type="bibr" rid="B32">32</xref>] employs a dual-path structure with a ResNet encoder, combining ResNet&#x2019;s feature extraction capabilities with U-Net&#x2019;s segmentation advantages, offering an alternative effective solution for medical image segmentation. Zhao et al. [<xref ref-type="bibr" rid="B33">33</xref>] proposed a four-layer feature calibration branch based on an attention mechanism. The downsampling layer reduces the resolution of rectal cancer CT image feature maps to half of the original size, followed by pointwise convolution to enable interactions between channels. This method effectively expands the receptive field of subsequent convolutional layers and optimizes computational efficiency by reducing the cost of calculating spatial attention. Model compression, as another approach to simplifying network structures, removes structural redundancy while maintaining performance, making it more suitable for various applications in medical image analysis. Wang et al. [<xref ref-type="bibr" rid="B34">34</xref>] designed a sophisticated teacher network to learn multi-scale features, guiding a more lightweight student network to improve segmentation accuracy. Experiments showed that this method effectively acquires detailed morphological features of the brain from the teacher network. Hajabdollahi et al. [<xref ref-type="bibr" rid="B35">35</xref>] proposed a channel pruning algorithm for medical image segmentation tasks, which selects color channels during image processing and allows training of the target structure directly on the pre-selected key channels. However, these studies did not address how to utilize the redundancy effectively.</p>
<p>Based on the above research findings, this paper proposes a lightweight segmentation model that emphasizes spatial and channel features. This model improves segmentation accuracy and efficiency without increasing additional computational costs, providing a new and efficient solution for the medical imaging field.</p>
</sec>
</sec>
<sec sec-type="methods" id="s3">
<title>Methods</title>
<sec id="s3-1">
<title>The overall framework of MCA-Net</title>
<p>As illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>, the proposed model framework consists primarily of the ISDConv module, the AEAM module, and dynamic convolution. The ISDConv module is composed of three parts: ScConv, Inception convolution, and standard convolution. By incorporating depthwise separable convolutions and group convolutions, ISDConv facilitates the model&#x2019;s understanding of multi-scale information within images, thereby enhancing its ability to detect and classify objects of varying sizes.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The overall framework of MCA-Net.</p>
</caption>
<graphic xlink:href="fphy-12-1532638-g001.tif"/>
</fig>
<p>The AEAM module operates in two stages: SEattention and SCSA. SEattention enhances the network&#x2019;s representational capacity by explicitly modeling the interdependencies between convolutional feature channels. SCSA, in turn, is divided into two components: SMSA and PCSA. SMSA integrates multi-semantic information and employs a progressive compression strategy to inject discriminative spatial priors into the channel self-attention mechanism of PCSA, effectively guiding channel recalibration. Within PCSA, robust feature interaction based on a self-attention mechanism further mitigates the multi-semantic information discrepancy among sub-features in SMSA.</p>
</sec>
<sec id="s3-2">
<title>Inception-Split depth convolution</title>
<p>As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, ISDConv consists of a ScConv, an Inception Convolution, and a standard Conv2d layer. The Inception Convolution achieves lightweight performance by efficiently decomposing a large kernel depthwise convolution into four parallel branches along the channel dimension. These branches consist of a small square kernel, two orthogonal large kernels, and an identity mapping. The use of a small square kernel reduces computational complexity, while the orthogonal large kernels capture different spatial information at varying scales. The identity mapping helps preserve the original input features, further enhancing the efficiency of the network. Additionally, this architecture incorporates 1 &#xd7; 1 convolutions for dimensionality reduction before applying computationally expensive operations, minimizing the computational burden while preserving the model&#x2019;s ability to learn rich, multi-scale features. These four branches not only achieve higher computational efficiency than the large kernel depthwise convolution but also maintain a large receptive field, enabling the model to capture spatial context effectively for improved performance.</p>
<p>One of the branches employs a <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> kernel, which avoids the inefficiency of large square kernels. Instead, large square kernels <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are decomposed into <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, significantly reducing computational complexity. Specifically, for a given input <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, it is divided into four groups along the channel dimension, with the operation defined as <xref ref-type="disp-formula" rid="e1">Equation 1</xref>:<disp-formula id="e1">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>:</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>:</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>:</mml:mo>
<mml:mi>g</mml:mi>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>g</mml:mi>
<mml:mo>:</mml:mo>
<mml:mn>3</mml:mn>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>:</mml:mo>
<mml:mn>3</mml:mn>
<mml:mi>g</mml:mi>
<mml:mo>:</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where, <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of channels in each convolution branch, which is determined by the formula <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the ratio for splitting and <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of input channels. The input is divided into four groups along the channel dimension based on this ratio, and the resulting split inputs are then fed into the respective parallel branches. Therefore, the following <xref ref-type="disp-formula" rid="e2">Equation 2</xref> can be established:<disp-formula id="e2">
<mml:math id="m11">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf10">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the <inline-formula id="inf11">
<mml:math id="m13">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> kernel size, <inline-formula id="inf12">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the kernel sizes of <inline-formula id="inf13">
<mml:math id="m15">
<mml:mrow>
<mml:mn>11</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>11</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the feature map, <inline-formula id="inf16">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> refers to the features in the width direction, and <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> refers to the features in the height dimension of the image. After processing each input <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> through its respective branch, the outputs <inline-formula id="inf19">
<mml:math id="m21">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are concatenated along the channel dimension. The operation can be expressed as <xref ref-type="disp-formula" rid="e3">Equation 3</xref>.<disp-formula id="e3">
<mml:math id="m22">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-3">
<title>Adaptive Enhanced Attention Module</title>
<p>This paper introduces the AEAM attention module, designed to achieve synergy between channel attention and spatial attention through dimensional decoupling, lightweight multi-semantic guidance, and semantic discrepancy mitigation. As shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, the AEAM module consists of two main components: SEattention and SCSA.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The overall framework of AEAM. SCSA uses multi-semantic spatial information to guide the learning of channel-wise self-attention. <inline-formula id="inf20">
<mml:math id="m23">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the batch size, <inline-formula id="inf21">
<mml:math id="m24">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> signifies the number of channels, and <inline-formula id="inf22">
<mml:math id="m25">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf23">
<mml:math id="m26">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> correspond to the height and width of the feature maps, respectively. The variable <inline-formula id="inf24">
<mml:math id="m27">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of groups into which sub-features are divided, and <inline-formula id="inf25">
<mml:math id="m28">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes a single pixel.</p>
</caption>
<graphic xlink:href="fphy-12-1532638-g002.tif"/>
</fig>
<p>The SCSA module is composed of two sequentially linked components: Shared Multi-Semantic Spatial Attention (SMSA) and Progressive Channel Self-Attention (PCSA). SMSA employs multi-scale, depth-sharing one-dimensional convolutions to extract spatial information at different semantic levels from four independent sub-features. This approach enables the efficient integration of diverse spatial semantics across sub-features. After SMSA modulates the feature maps, the resulting features are passed to PCSA. This component combines a progressive compression strategy with a channel-specific self-attention mechanism (CSA) to refine the feature representation further.</p>
<p>In this paper, a given input <inline-formula id="inf26">
<mml:math id="m29">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is applied global average pooling along the height and width dimensions to create two unidirectional 1D sequence structures: <inline-formula id="inf27">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf28">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. To learn diverse spatial distributions and contextual relationships, the feature set is divided into <inline-formula id="inf29">
<mml:math id="m32">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> equally sized and independent sub-features, such that <inline-formula id="inf30">
<mml:math id="m33">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf31">
<mml:math id="m34">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, each sub-feature has a channel count of <inline-formula id="inf32">
<mml:math id="m35">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf33">
<mml:math id="m36">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of channels in the original feature set. In this study, we set the default value <inline-formula id="inf34">
<mml:math id="m37">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, decomposing the features into <inline-formula id="inf35">
<mml:math id="m38">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-dimensional and <inline-formula id="inf36">
<mml:math id="m39">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-dimensional sub-features. During the decomposition process, 1D convolution is applied to each sub-feature. We employ lightweight shared convolutions for alignment, which implicitly model feature consistency across both dimensions by learning correlations.</p>
<p>The ablation formula is shown in <xref ref-type="disp-formula" rid="e4">Equation 4</xref>:<disp-formula id="e4">
<mml:math id="m40">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mn>1</mml:mn>
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2192;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mn>1</mml:mn>
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2192;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>Where <inline-formula id="inf37">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf38">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent feature maps in height and width dimensions respectively. SEattention introduces the &#x201c;Squeeze-and-Excitation&#x201d; (SE) block, which enhances the network&#x2019;s representational capacity by explicitly modeling the interdependencies between convolutional feature channels. The SE block employs a special mechanism that enables the network to perform feature recalibration. Through this mechanism, the block learns to selectively emphasize informative features while suppressing less useful ones by leveraging global information.</p>
<p>The structure of the SE block is illustrated in the lower part of <xref ref-type="fig" rid="F2">Figure 2</xref>. For any given transformation <inline-formula id="inf39">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which maps the input <inline-formula id="inf40">
<mml:math id="m44">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to a feature map <inline-formula id="inf41">
<mml:math id="m45">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which <inline-formula id="inf42">
<mml:math id="m46">
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, a corresponding SE block can be constructed to perform feature recalibration. The feature map <inline-formula id="inf43">
<mml:math id="m47">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> first undergoes a squeeze operation, which aggregates the feature map across the spatial dimensions to generate a channel descriptor. The function of this descriptor is to embed the global distribution of channel feature responses, thereby enabling all layers of the network to utilize information from the global receptive field. After the aggregation, an excitation operation follows. This operation, in the form of a simple self-gating mechanism, takes the embedding as input and generates a set of modulation weights for each channel. These weights are applied to the feature map <inline-formula id="inf44">
<mml:math id="m48">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to produce the output of the SE block, which can then be directly fed into subsequent layers of the network.</p>
</sec>
<sec id="s3-4">
<title>The loss function</title>
<p>In this study, each image in the dataset is associated with a corresponding binary mask. Skin lesion segmentation is treated as a pixel-level binary classification task, distinguishing the skin lesions from the background. The combination of Binary Cross-Entropy (BCE) loss and the Dice Similarity Coefficient (DSC) loss is used as the loss function to optimize the network parameters. This approach effectively addresses the challenge of skin lesion segmentation by balancing pixel accuracy and overlap between the predicted and ground truth masks.</p>
<p>The loss function, referred to as the BceDice loss, can be expressed as <xref ref-type="disp-formula" rid="e5">Equation 5</xref>:<disp-formula id="e5">
<mml:math id="m49">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">BCE</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">Dice</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mo>&#x2b;</mml:mo>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:mfrac>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">BCEDice</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">BCE</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">Dice</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf45">
<mml:math id="m50">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of samples, <inline-formula id="inf46">
<mml:math id="m51">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the ground truth label, <inline-formula id="inf47">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the predicted values, <inline-formula id="inf48">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the true label of sample <inline-formula id="inf49">
<mml:math id="m54">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf50">
<mml:math id="m55">
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf51">
<mml:math id="m56">
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denote the ground truth and the intersection of the predicted region, respectively. <inline-formula id="inf52">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf53">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the weights of the two loss functions. In this study, both weights are set to 1 by default.</p>
</sec>
</sec>
<sec id="s4">
<title>Experiment</title>
<sec id="s4-1">
<title>Datasets</title>
<p>The ISIC (International Skin Imaging Collaboration) datasets are benchmark datasets widely used in medical image analysis, particularly for dermoscopic image segmentation, classification, and automated skin cancer detection. These datasets feature high-resolution dermoscopic images with comprehensive annotations, including lesion boundaries, diagnostic labels, and metadata. Covering a diverse range of skin conditions, they are designed to support tasks such as lesion segmentation, feature extraction, and disease classification. Notably, the ISIC2017 and ISIC2018 datasets have been instrumental in advancing research on melanoma detection and other skin diseases through the annual ISIC Challenges. Our research is specifically conducted on the ISIC2017 and ISIC2018 datasets. <xref ref-type="fig" rid="F3">Figure 3</xref> are some sample images from the ISIC2017 and ISIC2018 datasets.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Examples of original images and their ground truth annotations from the ISIC2017 and ISIC2018 datasets.</p>
</caption>
<graphic xlink:href="fphy-12-1532638-g003.tif"/>
</fig>
</sec>
<sec id="s4-2">
<title>Experiment details</title>
<p>All experiments were implemented using the PyTorch framework and performed on a laptop equipped with an NVIDIA GeForce RTX 3080 Ti GPU with 8 GB of memory. Based on established practices, all images were normalized and resized to 256 <inline-formula id="inf54">
<mml:math id="m59">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 256 pixels. Data augmentation techniques, including vertical flipping, horizontal flipping, and random rotations, were applied. The loss function used was the BCE-Dice loss, as defined in <xref ref-type="disp-formula" rid="e6">Equation 6</xref>.<disp-formula id="e6">
<mml:math id="m60">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">BCE&#x2212;Dice</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mspace width="2em"/>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x22c5;</mml:mo>
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf55">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the ground truth label, <inline-formula id="inf56">
<mml:math id="m62">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> denotes the predicted value, <inline-formula id="inf57">
<mml:math id="m63">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of pixels, <inline-formula id="inf58">
<mml:math id="m64">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a small constant which is set to 10 in this work, <inline-formula id="inf59">
<mml:math id="m65">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf60">
<mml:math id="m66">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are the weights for the BCE and Dice components. AdamW was utilized as the optimizer with an initial learning rate of 0.001, dynamically adjusted using a cosine annealing scheduler. The maximum number of iterations was set to 50, with a minimum learning rate of 0.0001. The training process was conducted over 300 epochs with a batch size of 8.</p>
</sec>
<sec id="s4-3">
<title>Evaluation metrics</title>
<p>In this study, segmentation performance is assessed using the mean Intersection over Union (mIoU), Dice Similarity Coefficient (DSC), and Accuracy (Acc), as defined in <xref ref-type="disp-formula" rid="e7">Equation 7</xref>. Additionally, the number of parameters is represented by Params, measured in millions (M), and computational complexity is quantified in GFLOPs. It is important to note that both Params and GFLOPs are calculated based on an input size of <inline-formula id="inf61">
<mml:math id="m67">
<mml:mrow>
<mml:mn>256</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.<disp-formula id="e7">
<mml:math id="m68">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>m</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>D</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>Where, TP, FP, FN, and TN represent True Positives, False Positives, False Negatives, and True Negatives, respectively.</p>
</sec>
<sec id="s4-4">
<title>Segmentation result analysis</title>
<p>In this section, we conducted comparative experiments on melanoma segmentation using the ISIC2017 and ISIC2018 skin lesion segmentation datasets and evaluated the test results. The evaluation metrics include DSC, mIoU, params, and GFLOPs. The results are presented in <xref ref-type="table" rid="T1">Tables 1</xref>, <xref ref-type="table" rid="T2">2</xref>, where we perform a comprehensive comparison of the proposed model with the following methods: UNet [<xref ref-type="bibr" rid="B36">36</xref>], Transfuse [<xref ref-type="bibr" rid="B37">37</xref>], FATNet [<xref ref-type="bibr" rid="B38">38</xref>], MALUNet [<xref ref-type="bibr" rid="B39">39</xref>], QGD-Net [<xref ref-type="bibr" rid="B40">40</xref>], LCA-UNet [<xref ref-type="bibr" rid="B41">41</xref>], SCSONet [<xref ref-type="bibr" rid="B42">42</xref>], PL-Net [<xref ref-type="bibr" rid="B43">43</xref>], UCM-Net [<xref ref-type="bibr" rid="B44">44</xref>], CSAP-UNet-S [<xref ref-type="bibr" rid="B45">45</xref>], and ELA-Net [<xref ref-type="bibr" rid="B46">46</xref>].</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Experimental comparison of MCANet with other models on the ISIC2017 dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Params</th>
<th align="center">GFLOPs</th>
<th align="center">mIoU (%)</th>
<th align="center">DSC (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">UNet (2015) [<xref ref-type="bibr" rid="B36">36</xref>]</td>
<td align="center">7.77</td>
<td align="center">13.76</td>
<td align="center">76.98</td>
<td align="center">86.99</td>
</tr>
<tr>
<td align="center">TransFuse (2021) [<xref ref-type="bibr" rid="B37">37</xref>]</td>
<td align="center">26.16</td>
<td align="center">11.5</td>
<td align="center">79.21</td>
<td align="center">88.4</td>
</tr>
<tr>
<td align="center">FAT-Net (2022) [<xref ref-type="bibr" rid="B38">38</xref>]</td>
<td align="center">30</td>
<td align="center">23</td>
<td align="center">76.53</td>
<td align="center">85</td>
</tr>
<tr>
<td align="center">MALUNet (2022) [<xref ref-type="bibr" rid="B39">39</xref>]</td>
<td align="center">0.175</td>
<td align="center">0.083</td>
<td align="center">78.78</td>
<td align="center">88.13</td>
</tr>
<tr>
<td align="center">QGD-Net (2023) [<xref ref-type="bibr" rid="B40">40</xref>]</td>
<td align="center">0.777</td>
<td align="center">&#x2014;</td>
<td align="center">72.58</td>
<td align="center">84.1</td>
</tr>
<tr>
<td align="center">LCAUnet (2023) [<xref ref-type="bibr" rid="B41">41</xref>]</td>
<td align="center">13.38</td>
<td align="center">18.91</td>
<td align="center">76.1</td>
<td align="center">86.6</td>
</tr>
<tr>
<td align="center">SCSONet (2024) [<xref ref-type="bibr" rid="B42">42</xref>]</td>
<td align="center">0.149</td>
<td align="center">0.056</td>
<td align="center">80.14</td>
<td align="center">88.97</td>
</tr>
<tr>
<td align="center">PL-Net (2024) [<xref ref-type="bibr" rid="B43">43</xref>]</td>
<td align="center">15.03</td>
<td align="center">&#x2014;</td>
<td align="center">77.9</td>
<td align="center">85.9</td>
</tr>
<tr>
<td align="center">UCM-Net (2024) [<xref ref-type="bibr" rid="B44">44</xref>]</td>
<td align="center">0.499</td>
<td align="center">0.047</td>
<td align="center">80.71</td>
<td align="center">87.66</td>
</tr>
<tr>
<td align="center">CSAP-UNet-S (2024) [<xref ref-type="bibr" rid="B45">45</xref>]</td>
<td align="center">27.5</td>
<td align="center">8.918</td>
<td align="center">81.5</td>
<td align="center">88.8</td>
</tr>
<tr>
<td align="center">ELANet (2024) [<xref ref-type="bibr" rid="B46">46</xref>]</td>
<td align="center">0.459</td>
<td align="center">8.43</td>
<td align="center">82.87</td>
<td align="center">90.6</td>
</tr>
<tr>
<td align="center">MCANet (ours)</td>
<td align="center">0.128</td>
<td align="center">0.022</td>
<td align="center">83.25</td>
<td align="center">90.86</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Experimental comparison of MCANet with other models on the ISIC2018 dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Params</th>
<th align="center">GFLOPs</th>
<th align="center">mIoU (%)</th>
<th align="center">DSC (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">UNet (2015) [<xref ref-type="bibr" rid="B36">36</xref>]</td>
<td align="center">7.77</td>
<td align="center">13.76</td>
<td align="center">78.13</td>
<td align="center">86.99</td>
</tr>
<tr>
<td align="center">Unet &#x2b;&#x2b; (2018) [<xref ref-type="bibr" rid="B47">47</xref>]</td>
<td align="center">9.16</td>
<td align="center">34.86</td>
<td align="center">78.92</td>
<td align="center">87.83</td>
</tr>
<tr>
<td align="center">TransFuse (2021) [<xref ref-type="bibr" rid="B37">37</xref>]</td>
<td align="center">26.16</td>
<td align="center">11.5</td>
<td align="center">80.63</td>
<td align="center">89.27</td>
</tr>
<tr>
<td align="center">MALUNet (2022) [<xref ref-type="bibr" rid="B39">39</xref>]</td>
<td align="center">0.175</td>
<td align="center">0.083</td>
<td align="center">80.25</td>
<td align="center">89.04</td>
</tr>
<tr>
<td align="center">AMCC-Net (2023) [<xref ref-type="bibr" rid="B48">48</xref>]</td>
<td align="center">0.845</td>
<td align="center">&#x2014;</td>
<td align="center">80.18</td>
<td align="center">89</td>
</tr>
<tr>
<td align="center">SCSONet (2024) [<xref ref-type="bibr" rid="B42">42</xref>]</td>
<td align="center">0.149</td>
<td align="center">0.056</td>
<td align="center">80.99</td>
<td align="center">89.5</td>
</tr>
<tr>
<td align="center">MCNMF-Unet (2024) [<xref ref-type="bibr" rid="B49">49</xref>]</td>
<td align="center">0.332</td>
<td align="center">0.0538</td>
<td align="center">81.99</td>
<td align="center">89.96</td>
</tr>
<tr>
<td align="center">GIVTED-Net (2024) [<xref ref-type="bibr" rid="B50">50</xref>]</td>
<td align="center">0.19</td>
<td align="center">0.56</td>
<td align="center">79.79</td>
<td align="center">87.61</td>
</tr>
<tr>
<td align="center">UCM-Net (2024) [<xref ref-type="bibr" rid="B44">44</xref>]</td>
<td align="center">0.499</td>
<td align="center">0.047</td>
<td align="center">81.26</td>
<td align="center">88.48</td>
</tr>
<tr>
<td align="center">ELANet (2024) [<xref ref-type="bibr" rid="B46">46</xref>]</td>
<td align="center">0.459</td>
<td align="center">8.43</td>
<td align="center">81.85</td>
<td align="center">90.1</td>
</tr>
<tr>
<td align="center">MCANet (ours)</td>
<td align="center">0.128</td>
<td align="center">0.024</td>
<td align="center">83.68</td>
<td align="center">91.12</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In addition, bar charts are utilized in this study to visually illustrate the performance of different models on various metrics, providing a clearer comparison between our method and others. Specifically, for the comparison of lightweight metrics, only models designed with lightweight objectives were selected, with the results presented in <xref ref-type="fig" rid="F4">Figures 4</xref>, <xref ref-type="fig" rid="F5">5</xref>. The experimental results indicate that MCANet outperforms all other methods in both data sets in terms of DSC and mIoU metrics. Notably, MCANet achieves Dice scores exceeding 0.9 on the ISIC datasets, significantly outperforming all comparison models and demonstrating its superior segmentation performance.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Performance comparison of different models across various metrics on the ISIC2017 dataset.</p>
</caption>
<graphic xlink:href="fphy-12-1532638-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Performance comparison of different models across various metrics on the ISIC2018 dataset.</p>
</caption>
<graphic xlink:href="fphy-12-1532638-g005.tif"/>
</fig>
<p>Furthermore, to further validate the segmentation performance of the model, we present the visual segmentation results on the ISIC dataset, as shown in <xref ref-type="fig" rid="F6">Figures 6</xref>, <xref ref-type="fig" rid="F7">7</xref>. Although there are some differences between the MCANet segmentation results and mask images, MCANet outperforms other models in capturing detailed information from medical images, giving it a significant advantage in accurately segmenting the areas of the injury. Specifically, <xref ref-type="fig" rid="F8">Figure 8</xref> shows that MCANet can more accurately capture the target location in segmentation tasks involving smaller lesions, with finer and more precise segmentation of the lesion boundaries.However, our study also has some limitations. First, although MCANet demonstrates impressive performance on the ISIC datasets, its generalizability to other medical imaging datasets remains to be fully explored. In addition, while the model is lightweight in design, further optimization is required to meet the strict deployment constraints of resource-constrained devices, such as smartphones or embedded systems. Another limitation lies in the annotation quality of the datasets used, as potential noise in the segmentation masks may influence the model&#x2019;s learning process. Finally, despite MCANet&#x2019;s ability to capture detailed features, there are still some challenges in handling highly irregular or extremely small lesions, which may require more advanced attention mechanisms.To address these issues, future research will focus on several directions. First, extending the evaluation to additional datasets with diverse imaging modalities can help assess the robustness and versatility of MCANet. Second, incorporating techniques such as knowledge distillation or pruning could further improve the model&#x2019;s efficiency for deployment in real-time scenarios. Third, exploring semi-supervised or unsupervised learning methods may reduce dependency on high-quality annotations, enabling better performance even with noisy labels. Finally, integrating advanced multi-scale feature extraction modules could enhance the model&#x2019;s ability to handle challenging segmentation tasks involving complex lesion patterns.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Visual comparison of segmentation results of MCANet and other methods on the ISIC2017 dataset.</p>
</caption>
<graphic xlink:href="fphy-12-1532638-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Visual comparison of segmentation results of MCANet and other methods on the ISIC2018 dataset.</p>
</caption>
<graphic xlink:href="fphy-12-1532638-g007.tif"/>
</fig>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Heatmap visualization of melanoma lesion area segmentation.</p>
</caption>
<graphic xlink:href="fphy-12-1532638-g008.tif"/>
</fig>
</sec>
<sec id="s4-5">
<title>Ablation study on module effectiveness</title>
<p>To evaluate the contribution of each module in MCANet, we designed and conducted a series of ablation studies, with the results summarized in <xref ref-type="table" rid="T3">Table 3</xref>. Using the SCSONet baseline model as a reference, we performed comparative experiments with different combinations of the proposed modules on the ISIC dataset. Furthermore, to provide a clearer visualization of the impact of each module on segmentation performance, we used bar charts to illustrate variations in key metrics, such as DSC and mIoU, as shown in <xref ref-type="fig" rid="F9">Figure 9</xref>.In the ablation study, &#x201c;Base &#x2b; AEAM&#x201d; represents the integration of the proposed AEAM module into the baseline model, &#x201c;Base &#x2b; ISDConv&#x201d; denotes the addition of the ISDConv module to the baseline, and &#x201c;MCANet&#x201d; refers to the complete network architecture proposed in this study. From <xref ref-type="table" rid="T3">Table 3</xref> and the bar chart, it can be observed that integrating the proposed modules into the baseline model not only results in negligible increases in parameter count and computational complexity but also leads to significant improvements in segmentation performance. Specifically, as the modules are progressively added, the segmentation performance steadily improves, with the key metrics DSC and mIoU ultimately reaching 0.9086 and 0.8325, representing increases of <inline-formula id="inf62">
<mml:math id="m69">
<mml:mrow>
<mml:mn>2.93</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf63">
<mml:math id="m70">
<mml:mrow>
<mml:mn>5.37</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, respectively, compared to the baseline. The bar chart further illustrates this performance improvement trend, visually highlighting the contribution of each module.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Ablation experiments with different module combinations.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Params</th>
<th align="center">GFLOPs</th>
<th align="center">mIoU (%)</th>
<th align="center">DSC (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Base</td>
<td align="center">0.112</td>
<td align="center">0.021</td>
<td align="center">79.01</td>
<td align="center">88.27</td>
</tr>
<tr>
<td align="center">Base &#x2b; AEAM</td>
<td align="center">0.127</td>
<td align="center">0.022</td>
<td align="center">81.39</td>
<td align="center">90.34</td>
</tr>
<tr>
<td align="center">BASE &#x2b; ISDConv</td>
<td align="center">0.114</td>
<td align="center">0.022</td>
<td align="center">82.32</td>
<td align="center">90.84</td>
</tr>
<tr>
<td align="center">MCANet</td>
<td align="center">0.128</td>
<td align="center">0.024</td>
<td align="center">83.25</td>
<td align="center">90.86</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Visual representation of the impact of different modules on model performance.</p>
</caption>
<graphic xlink:href="fphy-12-1532638-g009.tif"/>
</fig>
<p>Moreover, the experimental results demonstrate that the proposed modules collaborate effectively, with the addition of individual modules not causing any degradation in overall performance but instead continuously improving segmentation accuracy. Additionally, our module design is highly adaptable, allowing for seamless integration into other network architectures without requiring significant modifications to the original structure. For instance, incorporating the AEAM or ISDConv modules into other networks results in varying degrees of performance improvement, validating the generalizability and practicality of the proposed modules.</p>
<p>In summary, the results of the ablation studies and their visual analysis demonstrate the significant contributions of the proposed modules to the model&#x2019;s performance. These improvements not only enhance the segmentation capability of MCANet but also highlight the academic significance and practical applicability of our work in the field of medical image segmentation.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>Conclusion</title>
<p>Medical image analysis typically requires significant computational resources, which directly impact diagnostic speed and accuracy. Advanced methods like deep learning are resource-intensive, making them difficult to implement in resource-constrained environments. To address this, we propose MCAN, a novel lightweight network architecture featuring ISDConv, AEAM, and dynamic convolution. Our model reduces computational costs while maintaining performance, achieving competitive segmentation with 0.128M parameters and 0.022 GFLOPs. However, due to the limited dataset, the model&#x2019;s generalization ability requires further investigation.</p>
<p>Future research can focus on several key areas. Firstly, further optimization of lightweight techniques and attention mechanisms is needed, especially for specific types of medical images. For example, improving the prediction accuracy and robustness of melanoma images across different skin types is an important direction. Additionally, due to the limited dataset size in this study, further validation of the model&#x2019;s generalization ability is required. Future work should aim to expand the dataset with more representative clinical data to assess the model&#x2019;s performance in real-world clinical environments, particularly in resource-constrained settings such as mobile medical devices or low-resource hospitals. Finally, our method could be extended to multi-modal tasks, such as integrated diagnosis using CT and MRI, with a focus on improving the model&#x2019;s fusion capability while maintaining computational efficiency.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>ZL: Conceptualization, Data curation, Investigation, Methodology, Project administration, Software, Supervision, Validation, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing. HW: Conceptualization, Formal Analysis, Resources, Software, Supervision, Validation, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing. HC: Data curation, Investigation, Methodology, Project administration, Resources, Supervision, Writing&#x2013;review and editing. CL: Conceptualization, Formal Analysis, Resources, Software, Writing&#x2013;review and editing. AY: Data curation, Formal Analysis, Funding acquisition, Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research was sponsored by the Special key project of Chongqing technology innovation and application development (CSTB2024TIAD-STX0023, CSTB2024TIAD-STX0030, CSTB2024TIAD-STX0037), Science and Technology Research Program of Chongqing Municipal Education (KJQN202400618) and &#x201c;Unveiling and Leading&#x201d; Project by the Chongqing Municipal Bureau of Industry and Information Technology (2022-37).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schadendorf</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Van Akkooi</surname>
<given-names>AC</given-names>
</name>
<name>
<surname>Berking</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Griewank</surname>
<given-names>KG</given-names>
</name>
<name>
<surname>Gutzmer</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Hauschild</surname>
<given-names>A</given-names>
</name>
<etal/>
</person-group> <article-title>Melanoma</article-title>. <source>The Lancet</source> (<year>2018</year>) <volume>392</volume>:<fpage>971</fpage>&#x2013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1016/s0140-6736(18)31559-9</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Tao</surname>
<given-names>D</given-names>
</name>
</person-group>. <article-title>Prototype-driven and multi-expert integrated multi-modal mr brain tumor image segmentation</article-title>. <source>IEEE Trans Instrumentation Meas</source> (<year>2024</year>) <volume>74</volume>:<fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1109/tim.2024.3500067</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Chai</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Deformation-aware and reconstruction-driven multimodal representation learning for brain tumor segmentation with missing modalities</article-title>. <source>Biomed Signal Process Control</source> (<year>2024</year>) <volume>91</volume>:<fpage>106012</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2024.106012</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Hua</surname>
<given-names>Z</given-names>
</name>
</person-group>. <article-title>Transformer-based multi-attention hybrid networks for skin lesion segmentation</article-title>. <source>Expert Syst Appl</source> (<year>2024</year>) <volume>244</volume>:<fpage>123016</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2023.123016</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Lian</surname>
<given-names>C</given-names>
</name>
</person-group>. <article-title>Msca-net: multi-scale contextual attention network for skin lesion segmentation</article-title>. <source>Pattern Recognition</source> (<year>2023</year>) <volume>139</volume>:<fpage>109524</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2023.109524</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zuo</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>A</given-names>
</name>
</person-group>. <article-title>Gfanet: gated fusion attention network for skin lesion segmentation</article-title>. <source>Comput Biol Med</source> (<year>2023</year>) <volume>155</volume>:<fpage>106462</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106462</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q</given-names>
</name>
<etal/>
</person-group> <article-title>X-net: brain stroke lesion segmentation based on depthwise separable convolution and long-range dependencies</article-title>. In: <conf-name>Proceedings, Part III 22nd International Conference Medical Image Computing and Computer Assisted Intervention&#x2013;MICCAI 2019</conf-name>; <conf-date>October 13&#x2013;17, 2019</conf-date>; <conf-loc>Shenzhen, China</conf-loc>. <publisher-name>Springer</publisher-name> (<year>2019</year>) p. <fpage>247</fpage>&#x2013;<lpage>55</lpage>.</citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X</given-names>
</name>
<etal/>
</person-group> <article-title>Msdf-net: multi-scale deep fusion network for stroke lesion segmentation</article-title>. <source>IEEE Access</source> (<year>2019</year>) <volume>7</volume>:<fpage>178486</fpage>&#x2013;<lpage>95</lpage>. <pub-id pub-id-type="doi">10.1109/access.2019.2958384</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Rema-net: an efficient multi-attention convolutional neural network for rapid skin lesion segmentation</article-title>. <source>Comput Biol Med</source> (<year>2023</year>) <volume>159</volume>:<fpage>106952</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.106952</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Mu</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X</given-names>
</name>
</person-group>. <article-title>Glioma segmentation-oriented multi-modal mr image fusion with adversarial learning</article-title>. <source>IEEE/CAA J Automatica Sinica</source> (<year>2022</year>) <volume>9</volume>:<fpage>1528</fpage>&#x2013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1109/jas.2022.105770</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Mazur</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Brain tumor segmentation in mri with multi-modality spatial information enhancement and boundary shape correction</article-title>. <source>Pattern Recognition</source> (<year>2024</year>) <volume>153</volume>:<fpage>110553</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2024.110553</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>He</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Cong</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Brain tumor segmentation based on the fusion of deep semantics and edge information in multimodal mri</article-title>. <source>Inf Fusion</source> (<year>2023</year>) <volume>91</volume>:<fpage>376</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1016/j.inffus.2022.10.022</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X</given-names>
</name>
</person-group>. <article-title>Rethinking the effectiveness of objective evaluation metrics in multi-focus image fusion: a statistic-based approach</article-title>. <source>IEEE Trans Pattern Anal Machine Intelligence</source> (<year>2024</year>) <volume>46</volume>:<fpage>5806</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1109/tpami.2024.3367905</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khan</surname>
<given-names>TM</given-names>
</name>
<name>
<surname>Naqvi</surname>
<given-names>SS</given-names>
</name>
<name>
<surname>Meijering</surname>
<given-names>E</given-names>
</name>
</person-group>. <article-title>Esdmr-net: a lightweight network with expand-squeeze and dual multiscale residual connections for medical image segmentation</article-title>. <source>Eng Appl Artif Intelligence</source> (<year>2024</year>) <volume>133</volume>:<fpage>107995</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2024.107995</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Nakagawa</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Shan</surname>
<given-names>X</given-names>
</name>
</person-group>. <article-title>A multi-attention and depthwise separable convolution network for medical image segmentation</article-title>. <source>Neurocomputing</source> (<year>2024</year>) <volume>564</volume>:<fpage>126970</fpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2023.126970</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z</given-names>
</name>
</person-group>. <article-title>LDCNet: limb direction cues-aware network for flexible HPE in industrial behavioral biometrics systems</article-title>. <source>IEEE Trans Ind Inform</source> (<year>2023</year>) <volume>20</volume>:<fpage>8068</fpage>&#x2013;<lpage>78</lpage>. <pub-id pub-id-type="doi">10.1109/tii.2023.3266366</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>F</given-names>
</name>
</person-group>. <article-title>Lmu-net: lightweight u-shaped network for medical image segmentation</article-title>. <source>Med and Biol Eng and Comput</source> (<year>2024</year>) <volume>62</volume>:<fpage>61</fpage>&#x2013;<lpage>70</lpage>. <pub-id pub-id-type="doi">10.1007/s11517-023-02908-w</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Pei</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>L</given-names>
</name>
<etal/>
</person-group> <article-title>Mlu-net: a multi-level lightweight u-net for medical image segmentation integrating frequency representation and mlp-based methods</article-title>. <source>IEEE Access</source> (<year>2024</year>) <volume>12</volume>:<fpage>20734</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1109/access.2024.3360889</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ruan</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Ege-unet: an efficient group enhanced unet for skin lesion segmentation</article-title>. In: <conf-name>International conference on medical image computing and computer-assisted intervention</conf-name>. <publisher-name>Springer</publisher-name> (<year>2023</year>) p. <fpage>481</fpage>&#x2013;<lpage>90</lpage>.</citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lei</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Nandi</surname>
<given-names>AK</given-names>
</name>
</person-group>. <article-title>Sgu-net: shape-guided ultralight network for abdominal image segmentation</article-title>. <source>IEEE J Biomed Health Inform</source> (<year>2023</year>) <volume>27</volume>:<fpage>1431</fpage>&#x2013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1109/jbhi.2023.3238183</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z</given-names>
</name>
</person-group>. <article-title>Dynamic convolution: attention over convolution kernels</article-title>. In: <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name> (<year>2020</year>) p. <fpage>11030</fpage>&#x2013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Cen</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z</given-names>
</name>
</person-group>. <article-title>Different input resolutions and arbitrary output resolution: a meta learning-based deep framework for infrared and visible image fusion</article-title>. <source>IEEE Trans Image Process</source> (<year>2021</year>) <volume>30</volume>:<fpage>4070</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1109/tip.2021.3069339</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>A deep learning framework for infrared and visible image fusion without strict registration</article-title>. <source>Int J Comp Vis</source> (<year>2024</year>) <volume>132</volume>:<fpage>1625</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-023-01948-x</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>Y</given-names>
</name>
<etal/>
</person-group> <article-title>Channel prior convolutional attention for medical image segmentation</article-title>. <source>Comput Biol Med</source> (<year>2024</year>) <volume>178</volume>:<fpage>108784</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.108784</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shaker</surname>
<given-names>AM</given-names>
</name>
<name>
<surname>Maaz</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Rasheed</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>MH</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>FS</given-names>
</name>
</person-group>. <article-title>Unetr&#x2b;&#x2b;: delving into efficient and accurate 3d medical image segmentation</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2024</year>). <pub-id pub-id-type="doi">10.1109/TMI.2024.3398728</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>Tsca-net: Transformer based spatial-channel attention segmentation network for medical images</article-title>. <source>Comput Biol Med</source> (<year>2024</year>) <volume>170</volume>:<fpage>107938</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.107938</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiong</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Zong</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Bian</surname>
<given-names>D</given-names>
</name>
<etal/>
</person-group> <article-title>Ina-net: an integrated noise-adaptive attention neural network for enhanced medical image segmentation</article-title>. <source>Expert Syst Appl</source> (<year>2024</year>) <volume>258</volume>:<fpage>125078</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2024.125078</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>E</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H</given-names>
</name>
</person-group>. <article-title>Combining external-latent attention for medical image segmentation</article-title>. <source>Neural Networks</source> (<year>2024</year>) <volume>170</volume>:<fpage>468</fpage>&#x2013;<lpage>77</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2023.10.046</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L</given-names>
</name>
</person-group>. <article-title>Medical image segmentation based on dynamic positioning and region-aware attention</article-title>. <source>Pattern Recognition</source> (<year>2024</year>) <volume>151</volume>:<fpage>110375</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2024.110375</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Pu</surname>
<given-names>L</given-names>
</name>
<etal/>
</person-group> <article-title>Ucunet: a lightweight and precise medical image segmentation network based on efficient large kernel u-shaped convolutional module design</article-title>. <source>Knowledge-Based Syst</source> (<year>2023</year>) <volume>278</volume>:<fpage>110868</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2023.110868</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Lan</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C</given-names>
</name>
<etal/>
</person-group> <article-title>Ucr-net: U-shaped context residual network for medical image segmentation</article-title>. <source>Comput Biol Med</source> (<year>2022</year>) <volume>151</volume>:<fpage>106203</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106203</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nisa</surname>
<given-names>SQ</given-names>
</name>
<name>
<surname>Ismail</surname>
<given-names>AR</given-names>
</name>
</person-group>. <article-title>Dual u-net with resnet encoder for segmentation of medical images</article-title>. <source>Int J Adv Comp Sci Appl</source> (<year>2022</year>) <volume>13</volume>. <pub-id pub-id-type="doi">10.14569/ijacsa.2022.0131265</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>W</given-names>
</name>
<etal/>
</person-group> <article-title>Efficient multi-organ segmentation from 3d abdominal ct images with lightweight network and knowledge distillation</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2023</year>) <volume>42</volume>:<fpage>2513</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1109/tmi.2023.3262680</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>D</given-names>
</name>
<etal/>
</person-group> <article-title>Segmenting neuronal structure in 3d optical microscope images via knowledge distillation with teacher-student network</article-title>. In: <source>2019 IEEE 16th international symposium on biomedical imaging (ISBI 2019)</source>. <publisher-name>IEEE</publisher-name> (<year>2019</year>) p. <fpage>228</fpage>&#x2013;<lpage>31</lpage>.</citation>
</ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hajabdollahi</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Esfandiarpoor</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Khadivi</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Soroushmehr</surname>
<given-names>SMR</given-names>
</name>
<name>
<surname>Karimi</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Samavi</surname>
<given-names>S</given-names>
</name>
</person-group>. <article-title>Simplification of neural networks for skin lesion image segmentation using color channel pruning</article-title>. <source>Comput Med Imaging Graphics</source> (<year>2020</year>) <volume>82</volume>:<fpage>101729</fpage>. <pub-id pub-id-type="doi">10.1016/j.compmedimag.2020.101729</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ronneberger</surname>
<given-names>O</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Brox</surname>
<given-names>T</given-names>
</name>
</person-group>. <article-title>U-net: convolutional networks for biomedical image segmentation</article-title>. In: <conf-name>proceedings, part III 18th international conference Medical image computing and computer-assisted intervention&#x2013;MICCAI 2015</conf-name>; <conf-date>October 5-9, 2015</conf-date>; <conf-loc>Munich, Germany</conf-loc>. <publisher-name>Springer</publisher-name> (<year>2015</year>) p. <fpage>234</fpage>&#x2013;<lpage>41</lpage>.</citation>
</ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Q</given-names>
</name>
</person-group>. <article-title>Transfuse: fusing transformers and cnns for medical image segmentation</article-title>. In: <conf-name>proceedings, Part I 24th international conference Medical image computing and computer assisted intervention&#x2013;MICCAI 2021</conf-name>; <conf-date>September 27&#x2013;October 1, 2021</conf-date>; <conf-loc>Strasbourg, France</conf-loc>. <publisher-name>Springer</publisher-name> (<year>2021</year>) p. <fpage>14</fpage>&#x2013;<lpage>24</lpage>.</citation>
</ref>
<ref id="B38">
<label>38.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>Z</given-names>
</name>
</person-group>. <article-title>Fat-net: feature adaptive transformers for automated skin lesion segmentation</article-title>. <source>Med image Anal</source> (<year>2022</year>) <volume>76</volume>:<fpage>102327</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2021.102327</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ruan</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Xiang</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Malunet: a multi-attention and light-weight unet for skin lesion segmentation</article-title>. In: <conf-name>2022 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name>. <publisher-name>IEEE</publisher-name> (<year>2022</year>) p. <fpage>1150</fpage>&#x2013;<lpage>6</lpage>.</citation>
</ref>
<ref id="B40">
<label>40.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Pun</surname>
<given-names>CM</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>Qgd-net: a lightweight model utilizing pixels of affinity in feature layer for dermoscopic lesion segmentation</article-title>. <source>IEEE J Biomed Health Inform</source> (<year>2023</year>) <volume>27</volume>:<fpage>5982</fpage>&#x2013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1109/jbhi.2023.3320953</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Fft pattern recognition of crystal hrtem image with deep learning</article-title>. <source>Micron</source> (<year>2023</year>) <volume>166</volume>:<fpage>103402</fpage>. <pub-id pub-id-type="doi">10.1016/j.micron.2022.103402</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>L</given-names>
</name>
<etal/>
</person-group> <article-title>Scsonet: spatial-channel synergistic optimization net for skin lesion segmentation</article-title>. <source>Front Phys</source> (<year>2024</year>) <volume>12</volume>:<fpage>1388364</fpage>. <pub-id pub-id-type="doi">10.3389/fphy.2024.1388364</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Ming</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>M</given-names>
</name>
</person-group>. <source>Pl-net: progressive learning network for medical image segmentation</source> (<year>2021</year>) <comment>arXiv preprint arXiv:2110.14484</comment>.</citation>
</ref>
<ref id="B44">
<label>44.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weng</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C</given-names>
</name>
</person-group>. <article-title>Ucm-net: a u-net-like tampered-region-related framework for copy-move forgery detection</article-title>. <source>IEEE Trans Multimedia</source> (<year>2023</year>) <volume>26</volume>:<fpage>750</fpage>&#x2013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.1109/tmm.2023.3270629</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Xin</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>L</given-names>
</name>
</person-group>. <article-title>Csap-unet: convolution and self-attention paralleling network for medical image segmentation with edge enhancement</article-title>. <source>Comput Biol Med</source> (<year>2024</year>) <volume>172</volume>:<fpage>108265</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.108265</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nie</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>S</given-names>
</name>
</person-group>. <article-title>Ela-net: an efficient lightweight attention network for skin lesion segmentation</article-title>. <source>Sensors</source> (<year>2024</year>) <volume>24</volume>:<fpage>4302</fpage>. <pub-id pub-id-type="doi">10.3390/s24134302</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Rahman Siddiquee</surname>
<given-names>MM</given-names>
</name>
<name>
<surname>Tajbakhsh</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>Unet&#x2b;&#x2b;: a nested u-net architecture for medical image segmentation</article-title>. In: <conf-name>Proceedings 4 Deep Learning in Medical Image Analysis and Multimodal Learning for Clinical Decision Support: 4th International Workshop, DLMIA 2018, and 8th International Workshop, ML-CDS 2018, Held in Conjunction with MICCAI 2018</conf-name>; <conf-date>September 20, 2018</conf-date>; <conf-loc>Granada, Spain</conf-loc>. <publisher-name>Springer</publisher-name> (<year>2018</year>) p. <fpage>3</fpage>&#x2013;<lpage>11</lpage>.</citation>
</ref>
<ref id="B48">
<label>48.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dayananda</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Yamanakkanavar</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>B</given-names>
</name>
</person-group>. <article-title>Amcc-net: an asymmetric multi-cross convolution for skin lesion segmentation on dermoscopic images</article-title>. <source>Eng Appl Artif Intelligence</source> (<year>2023</year>) <volume>122</volume>:<fpage>106154</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2023.106154</pub-id>
</citation>
</ref>
<ref id="B49">
<label>49.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Mcnmf-unet: a mixture conv-mlp network with multi-scale features fusion unet for medical image segmentation</article-title>. <source>PeerJ Comp Sci</source> (<year>2024</year>) <volume>10</volume>:<fpage>e1798</fpage>. <pub-id pub-id-type="doi">10.7717/peerj-cs.1798</pub-id>
</citation>
</ref>
<ref id="B50">
<label>50.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Fahsi</surname>
<given-names>RDH</given-names>
</name>
<name>
<surname>Prawirosoenoto</surname>
<given-names>ANF</given-names>
</name>
<name>
<surname>Nugroho</surname>
<given-names>HA</given-names>
</name>
<name>
<surname>Ardiyanto</surname>
<given-names>I</given-names>
</name>
</person-group>. <article-title>Givted-net: ghostnet-mobile involution vit encoder-decoder network for lightweight medical image segmentation</article-title>. <source>IEEE Access</source> (<year>2024</year>) <volume>12</volume>:<fpage>81281</fpage>&#x2013;<lpage>92</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2024.3411870</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>