<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Phys.</journal-id>
<journal-title>Frontiers in Physics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Phys.</abbrev-journal-title>
<issn pub-type="epub">2296-424X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1614983</article-id>
<article-id pub-id-type="doi">10.3389/fphy.2025.1614983</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>HiImp-SMI: an implicit transformer framework with high-frequency adapter for medical image segmentation</article-title>
<alt-title alt-title-type="left-running-head">Huang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphy.2025.1614983">10.3389/fphy.2025.1614983</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Lianchao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3031261/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Peng</surname>
<given-names>Feng</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3101200/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Binghao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3100704/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Cao</surname>
<given-names>Yinghong</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3101560/overview"/>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Information Science and Engineering</institution>, <institution>Dalian Polytechnic University</institution>, <addr-line>Dalian</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Information Technology Center</institution>, <institution>Dalian Polytechnic University</institution>, <addr-line>Dalian</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Biological Engineering</institution>, <institution>Dalian Polytechnic University</institution>, <addr-line>Dalian</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1995827/overview">Hairong Lin</ext-link>, Central South University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1598376/overview">Feifei Yang</ext-link>, Xi&#x2019;an University of Science and Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1742285/overview">Shuang Zhou</ext-link>, Chongqing Normal University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Feng Peng, <email>pengfeng@dlpu.edu.cn</email>; Yinghong Cao, <email>caoyinghong@dlpu.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>26</day>
<month>06</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>13</volume>
<elocation-id>1614983</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Huang, Peng, Huang and Cao.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Huang, Peng, Huang and Cao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Accurate and generalizable segmentation of medical images remains a challenging task due to boundary ambiguity and variations across domains. In this paper, an implicit transformer framework with a high-frequency adapter for medical image segmentation (HiImp-SMI) is proposed. A new dual-branch architecture is designed to simultaneously process spatial and frequency information, enhancing both boundary refinement and domain adaptability. Specifically, a Channel Attention Block selectively amplifies high-frequency boundary cues, improving contour delineation. A Multi-Branch Cross-Attention Block facilitates efficient hierarchical feature fusion, addressing challenges in multi-scale representation.Additionally, a ViT-Conv Fusion Block adaptively integrates global contextual awareness from Transformer features with local structural details, thereby significantly boosting cross-domain generalization. The entire network is trained in a supervised end-to-end manner, with frequency-adaptive modules integrated into the encoding stages of the Transformer backbone. Experimental evaluations show that HiImp-SMI consistently outperforms mainstream models on the Kvasir-Sessile and BCV datasets, including state-of-the-art implicit methods. For example, on the Kvasir-Sessile dataset, HiImp-SMI achieves a Dice score of 92.39%, outperforming I-MedSAM by 1%. On BCV, it demonstrates robust multi-class segmentation with consistent superiority across organs. These quantitative results demonstrate the framework&#x2019;s effectiveness in refining boundary precision, optimizing multi-scale feature representation, and improving cross-dataset generalization. This improvement is largely attributed to the dual-branch design and the integration of frequency-aware attention mechanisms, which enable the model to capture both anatomical details and domain-robust features. The proposed framework may serve as a flexible baseline for future work involving implicit modeling and multi-modal representation learning in medical image analysis.</p>
</abstract>
<kwd-group>
<kwd>nonlinear system</kwd>
<kwd>medical image segmentation</kwd>
<kwd>high-frequency adapter</kwd>
<kwd>cross-attention</kwd>
<kwd>feature fusion</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Interdisciplinary Physics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Medical image segmentation plays a crucial role in assisting disease diagnosis and guiding clinical treatment. Traditional discrete methods based on convolutional neural networks (CNNs), such as U-Net [<xref ref-type="bibr" rid="B1">1</xref>], nnU-Net [<xref ref-type="bibr" rid="B2">2</xref>], and PraNet [<xref ref-type="bibr" rid="B3">3</xref>], effectively integrate multi-scale features but remain highly sensitive to variations in data distribution, thus limiting cross-domain generalization. Although boundary-aware methods, such as Boundary-aware U-Net [<xref ref-type="bibr" rid="B4">4</xref>], WM-DOVA [<xref ref-type="bibr" rid="B5">5</xref>], Hausdorff distance-based approaches [<xref ref-type="bibr" rid="B6">6</xref>], dropout-based calibration [<xref ref-type="bibr" rid="B7">7</xref>], and neural network calibration [<xref ref-type="bibr" rid="B8">8</xref>], have improved localization precision and feature representation, these methods still face challenges when dealing with complex medical structures and achieving consistent segmentation performance across different domains. Additionally, multi-scale residual architectures like Res2Net [<xref ref-type="bibr" rid="B9">9</xref>] further enhance feature representation but are still limited in boundary preservation.</p>
<p>Recent developments have introduced Transformer-based architectures, such as TransUNet [<xref ref-type="bibr" rid="B10">10</xref>] and UNETR [<xref ref-type="bibr" rid="B11">11</xref>], leveraging global contextual awareness through self-attention mechanisms [<xref ref-type="bibr" rid="B12">12</xref>]. Despite superior global feature capture capabilities, these approaches often underperform in local boundary refinement and require extensive training data for effective generalization. Further advancements, such as LoRA [<xref ref-type="bibr" rid="B13">13</xref>], aim to improve Transformer efficiency and generalization but do not explicitly optimize for boundary segmentation accuracy. Furthermore, adaptations based on the Segment Anything Model (SAM) [<xref ref-type="bibr" rid="B14">14</xref>], including MedSAM [<xref ref-type="bibr" rid="B15">15</xref>], SAM-based 3D extensions [<xref ref-type="bibr" rid="B16">16</xref>], and customized SAM models [<xref ref-type="bibr" rid="B17">17</xref>], generally improve generalization capabilities but typically neglect fine-grained feature integration, resulting in limited boundary segmentation accuracy. Additional SAM-related studies, such as NTo3D [<xref ref-type="bibr" rid="B18">18</xref>], Customized SAM [<xref ref-type="bibr" rid="B19">19</xref>], SAM-Med2D [<xref ref-type="bibr" rid="B20">20</xref>], DiffDP [<xref ref-type="bibr" rid="B21">21</xref>], spatial prior-based approaches [<xref ref-type="bibr" rid="B22">22</xref>], and mask-enhanced SAM models [<xref ref-type="bibr" rid="B23">23</xref>], have explored further improvements but continue to face challenges with boundary precision.</p>
<p>Beyond conventional deep learning approaches, emerging research spans several interdisciplinary directions that address these challenges. For instance, memristor- and memcapacitor-based neural network models have been proposed to enable neuromorphic hardware implementations [<xref ref-type="bibr" rid="B24">24</xref>, <xref ref-type="bibr" rid="B25">25</xref>]; such analog in-memory circuits have demonstrated improved image segmentation speed and accuracy via parallel high-efficiency computations [<xref ref-type="bibr" rid="B26">26</xref>, <xref ref-type="bibr" rid="B27">27</xref>]. Recent studies have further explored Hamiltonian conservative chaotic systems integrated with memristors for modeling and FPGA implementation, enhancing the physical interpretability and stability of neuromorphic designs [<xref ref-type="bibr" rid="B28">28</xref>]. Similarly, chaotic and hyperchaotic dynamical systems have been exploited in image encryption, leveraging their high-dimensional unpredictability to enhance security. In particular, memristor-coupled cellular neural networks based on resonant tunneling diodes have been applied in forensic digital image protection, offering a secure hardware foundation for sensitive applications [<xref ref-type="bibr" rid="B29">29</xref>]. Some studies even integrate memristive chaotic circuits to strengthen resistance against differential attacks [<xref ref-type="bibr" rid="B30">30</xref>], and in general hyper-chaos offers greater randomness and key space than lower-dimensional maps [<xref ref-type="bibr" rid="B31">31</xref>], yielding encryption schemes with robust immunity to cryptanalytic attacks [<xref ref-type="bibr" rid="B32">32</xref>]. Other researchers have implemented novel hyperchaotic systems in FPGA to support audio encryption, demonstrating the practical deployment of such dynamics on low-power reconfigurable hardware [<xref ref-type="bibr" rid="B33">33</xref>, <xref ref-type="bibr" rid="B34">34</xref>]. In IoT contexts, researchers have developed lightweight image encryption and steganography techniques to secure multimedia data with minimal computational overhead [<xref ref-type="bibr" rid="B35">35</xref>, <xref ref-type="bibr" rid="B36">36</xref>], addressing the limitations of earlier cryptosystems on resource-constrained devices [<xref ref-type="bibr" rid="B37">37</xref>]. Moreover, discrete n-dimensional hyperchaotic maps with customizable Lyapunov exponents have been proposed to expand the design space for secure communications and embedded cryptography [<xref ref-type="bibr" rid="B38">38</xref>]. Additionally, integrating multi-modal information has become crucial for improving diagnostic accuracy, prompting new architectures that effectively fuse heterogeneous medical data streams [<xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B40">40</xref>]. Equally important, domain-generalization strategies are being pursued to ensure models remain robust across disparate imaging domains, tackling the severe performance degradation caused by cross-modality shifts without requiring retraining on target data [<xref ref-type="bibr" rid="B41">41</xref>]. Finally, a concerted effort is underway to translate these advances into practical deployments: specialized DSP-based accelerators and other hardware implementations are achieving real-time image processing with low power consumption [<xref ref-type="bibr" rid="B42">42</xref>, <xref ref-type="bibr" rid="B43">43</xref>], and even complex neuromorphic networks are being prototyped on DSP platforms [<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B26">26</xref>]. These developments across hardware design, secure encryption, lightweight algorithms, and multi-modal learning collectively strengthen the foundation for next-generation medical image segmentation systems.</p>
<p>Implicit neural representation methods represent another advancement, employing continuous mappings from coordinate spaces to representation spaces, exemplified by OSSNet [<xref ref-type="bibr" rid="B44">44</xref>], IOSNet [<xref ref-type="bibr" rid="B45">45</xref>], and SWIPE [<xref ref-type="bibr" rid="B46">46</xref>]. These models exhibit improved segmentation robustness across resolutions but remain constrained by their reliance on traditional convolutional encoders, limiting their capacity to simultaneously capture detailed boundary information and global contextual features. Further implicit methods, including NeRF [<xref ref-type="bibr" rid="B47">47</xref>], NUDF [<xref ref-type="bibr" rid="B48">48</xref>], NISF [<xref ref-type="bibr" rid="B49">49</xref>], ImplicitAtlas [<xref ref-type="bibr" rid="B50">50</xref>], implicit neural representations survey [<xref ref-type="bibr" rid="B51">51</xref>], shape reconstruction from sparse measurements [<xref ref-type="bibr" rid="B52">52</xref>], implicit functions for 3D reconstruction [<xref ref-type="bibr" rid="B53">53</xref>], MRI super-resolution [<xref ref-type="bibr" rid="B16">16</xref>], and volumetric SAM adaptations [<xref ref-type="bibr" rid="B54">54</xref>], have significant potential but share similar limitations. Frequency-domain adapters, like those in I-MedSAM [<xref ref-type="bibr" rid="B55">55</xref>], have enhanced boundary delineation, but single-adapter designs remain insufficient for comprehensive multi-scale feature integration.</p>
<p>To address these challenges, this study introduces HiImp-SMI, an implicit Transformer-based medical image segmentation framework incorporating three key innovations: (1) a Channel Attention Block to explicitly enhance high-frequency boundary information, (2) a Multi-Branch Cross-Attention Block to facilitate efficient hierarchical feature fusion across different scales, and (3) a ViT-Conv Fusion Block designed to integrate global context from Transformer-based architectures with local fine-grained features extracted by convolutional networks. Experimental validations conducted on the Kvasir-Sessile and BCV datasets demonstrate that HiImp-SMI outperforms existing segmentation methods, highlighting its effectiveness in boundary precision, multi-scale feature representation, and cross-dataset generalization capabilities.</p>
<p>The remainder of this paper is organized as follows: <xref ref-type="sec" rid="s2">Section 2</xref> details the proposed HiImp-SMI framework; <xref ref-type="sec" rid="s3">Section 3</xref> presents the experimental setup and results; and <xref ref-type="sec" rid="s4">Section 4</xref> concludes the study, providing directions for future research.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<p>The overall architecture of the proposed HiImp-SMI framework is depicted in <xref ref-type="fig" rid="F1">Figure 1</xref>. It comprises a dual-branch encoder structure that jointly exploits spatial-domain and frequency-domain information. Given an input image <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, a Fast Fourier Transform (FFT) is applied to derive its frequency representation <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>FFT</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which highlights high-frequency components corresponding to anatomical boundaries and texture transitions. By integrating <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>FFT</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> into the encoder, our Channel Attention Block can selectively amplify boundary-sensitive features, enhancing fine-grained localization and generalization to unseen domains. These embeddings are then processed by three key modules: a Channel Attention Block, which selectively enhances high-frequency boundary details; a Multi-Branch Cross Attention Block, designed to enable effective feature exchange across hierarchical levels; and a ViT-Conv Fusion Block, which adaptively integrates global contextual information from the Transformer branch and local structural features from the convolutional branch. Through this architecture, HiImp-SMI aims to achieve more precise boundary segmentation, stronger multi-scale representation, and enhanced cross-domain generalization.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Overall architecture of our proposed model.</p>
</caption>
<graphic xlink:href="fphy-13-1614983-g001.tif">
<alt-text content-type="machine-generated">Diagram of a medical image processing architecture. At the top, PDB Loss connects to an implicit decoder and a coordinate prompt, which links to a segment of a CT scan. The path flows through a prompt encoder to numerous ViT blocks with low-rank and frequency adapters, eventually arriving at a patch embedding stage. It continues to a Vit-conv fusion block, multi-branch cross attention block, and channel attention block. Coarse box prompt and FFT (Fast Fourier Transform) imagery feed into the medical image encoder. The diagram outlines each functional component's integration within this system.</alt-text>
</graphic>
</fig>
<sec id="s2-1">
<title>2.1 Channel attention block</title>
<p>In this study, SAM employs a Vision Transformer (ViT) as the image encoder, pretrained on a large-scale natural image dataset. To preserve the strong feature representation capability of the pretrained ViT, its weights are kept frozen during training. Instead, a local adapter module is introduced to incorporate localized inductive biases into the model, as illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The Channel Attention Block for domain-specific feature enhancement in the ViT encoder.</p>
</caption>
<graphic xlink:href="fphy-13-1614983-g002.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a neural network architecture. From bottom to top: Layer Normalization, Pointwise Convolution (PW Conv), Depthwise Convolution (DW Conv), Squeeze-and-Excitation Attention (SE Attention), another PW Conv, followed by a summation operation indicated by a circle with a plus sign. Arrows show the flow direction.</alt-text>
</graphic>
</fig>
<p>The Channel Attention Block enhances the domain-specific feature extraction capability of the pretrained Vision Transformer (ViT) without fine-tuning its weights. The procedure involves the following steps:<list list-type="simple">
<list-item>
<p>Step 1: Obtain the input embedding <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>vit</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from the ViT attention block. This embedding carries high-level semantic features. It serves as the input to the channel attention block.</p>
</list-item>
<list-item>
<p>Step 2: Apply layer normalization (LN) to stabilize feature distributions. LN normalizes each channel to reduce internal covariate shift. This improves training stability and convergence.</p>
</list-item>
<list-item>
<p>Step 3: Perform a pointwise convolution <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Conv</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> to adjust channel dimensions. This operation projects features into a latent space. It preserves spatial structure while enabling channel-wise transformation.</p>
</list-item>
<list-item>
<p>Step 4: Execute a depthwise convolution <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>DWConv</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> to capture spatial information. Each channel is convolved independently to extract local patterns. This enhances spatial modeling without increasing parameter count significantly.</p>
</list-item>
<list-item>
<p>Step 5: Apply a Squeeze-and-Excitation (SE) block to model channel-wise dependencies. Specifically, the SE block performs global average pooling followed by two fully connected layers and non-linear activations to generate a channel attention vector <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which is then applied to recalibrate the feature map, as shown in <xref ref-type="disp-formula" rid="e1">Equation 1</xref>:</p>
</list-item>
</list>
<disp-formula id="e1">
<mml:math id="m8">
<mml:mrow>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="cases">
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>z</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="left">
<mml:mtext>SE</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo>&#x2297;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mspace width="1em"/>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>Here, <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the input feature map, and <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the channel-wise descriptor obtained by global average pooling. <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are learnable weight matrices of two fully connected layers. <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denote the ReLU and sigmoid activation functions, respectively. The resulting attention vector <inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is used to rescale each channel of <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> via element-wise multiplication, enabling adaptive channel emphasis.<list list-type="simple">
<list-item>
<p>Step 6: Integrate the processed features using another pointwise convolution <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext>Conv</mml:mtext>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> to obtain refined embedding <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mtext>vit</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>, as defined in <xref ref-type="disp-formula" rid="e2">Equation 2</xref>:</p>
</list-item>
</list>
<disp-formula id="e2">
<mml:math id="m19">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mtext>vit</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Conv</mml:mtext>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mtext>SE</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mtext>DWConv</mml:mtext>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mtext>Conv</mml:mtext>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mtext>LN</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>vit</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<list list-type="simple">
<list-item>
<p>Step 7: Merge the refined features with the original features through a residual connection, as formulated in <xref ref-type="disp-formula" rid="e3">Equation 3</xref>:</p>
</list-item>
</list>
<disp-formula id="e3">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>out</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>vit</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mtext>vit</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-2">
<title>2.2 Multi-branch Cross Attention Block</title>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> illustrates the structure of the Multi-branch Cross Attention Block, which integrates deep features from the ViT branch with shallow features from a convolutional branch. The procedure involves the following steps:<list list-type="simple">
<list-item>
<p>Step 1: Extract shallow features <inline-formula id="inf18">
<mml:math id="m21">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> from the resized input image using a lightweight convolutional block. This step captures low-level visual patterns such as edges and textures. The convolutional block is designed to be efficient for early-stage feature extraction.</p>
</list-item>
<list-item>
<p>Step 2: Generate queries, keys, and values for the ViT branch and convolutional branch separately, as described in <xref ref-type="disp-formula" rid="e4">Equation 4</xref>:</p>
</list-item>
</list>
<disp-formula id="e4">
<mml:math id="m22">
<mml:mrow>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>Here, <inline-formula id="inf19">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf20">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote deep features from the ViT branch and shallow features from the convolutional branch, respectively. <inline-formula id="inf21">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents bottleneck features shared across branches. <inline-formula id="inf22">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf23">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf24">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are learnable linear projection matrices used to obtain queries <inline-formula id="inf25">
<mml:math id="m29">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, keys <inline-formula id="inf26">
<mml:math id="m30">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, and values <inline-formula id="inf27">
<mml:math id="m31">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> for attention computation.<list list-type="simple">
<list-item>
<p>Step 3: Fuse features across branches using deformable attention, detailed in <xref ref-type="disp-formula" rid="e5">Equation 5</xref>:</p>
</list-item>
</list>
<disp-formula id="e5">
<mml:math id="m32">
<mml:mrow>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>DeformAttn</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>DeformAttn</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>Here, <inline-formula id="inf28">
<mml:math id="m33">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf29">
<mml:math id="m34">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represent the cross-attended features refined via deformable attention in the ViT and convolutional branches, respectively. Deformable attention adaptively samples spatial locations, enabling the model to focus on semantically relevant regions. This mechanism facilitates more effective feature alignment across the two branches.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The Multi-branch Cross Attention Block for fusing ViT and convolutional features via cross-attention.</p>
</caption>
<graphic xlink:href="fphy-13-1614983-g003.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a neural network architecture with two parallel components. The left component, in blue, includes feed forward, layer normalization, and deformable attention operations, denoted as \(F_d^1\), \(F_d^c\), and \(Q, V_1, V_2\). The right component, in red, mirrors this structure, labeled with similar operations as \(F_s^1\), \(F_s^c\). Both components connect to color-coded inputs and outputs at the bottom labeled \(F_d\), \(F_b\), and \(F_s\), indicating data flow through the network.</alt-text>
</graphic>
</fig>
<p>
<list list-type="simple">
<list-item>
<p>Step 4: Refine the fused features with residual feedforward networks (FFN) and layer normalization (LN)&#x2014;this refinement is formalized in <xref ref-type="disp-formula" rid="e6">Equation 6</xref>:</p>
</list-item>
</list>
<disp-formula id="e6">
<mml:math id="m35">
<mml:mrow>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>FFN</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mtext>LN</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>FFN</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mtext>LN</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>Here, <inline-formula id="inf30">
<mml:math id="m36">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf31">
<mml:math id="m37">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> denote the updated deep and shallow features after refinement. The FFN enhances non-linear representation capacity, while LN improves training stability. The residual connection facilitates efficient information preservation and gradient flow.</p>
</sec>
<sec id="s2-3">
<title>2.3 ViT-Conv fusion block</title>
<p>A fusion block equipped with an automatic selection mechanism is constructed to integrate the diverse information provided by convolutional features and Transformer features. The architectural details of this module are illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The ViT-Conv Fusion Block for adaptive integration of Transformer and convolutional features.</p>
</caption>
<graphic xlink:href="fphy-13-1614983-g004.tif">
<alt-text content-type="machine-generated">Flowchart of a neural network process involving two parallel data paths. Fd and Fs inputs undergo transformations via Fully Connected (FC) and GELU layers. Outputs from both paths are combined using a Sigmoid function to obtain weights, &#x3C9; and 1-&#x3C9;. The weighted combinations are then merged to produce the final output, F_output.</alt-text>
</graphic>
</fig>
<p>The ViT-Conv Fusion Block adaptively integrates convolutional and Transformer features through these steps:<list list-type="simple">
<list-item>
<p>Step 1: Process deep <inline-formula id="inf32">
<mml:math id="m38">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and shallow <inline-formula id="inf33">
<mml:math id="m39">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> features individually with a channel attention layer to obtain logits <inline-formula id="inf34">
<mml:math id="m40">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. Channel attention highlights informative channels in each branch. This yields two attention logits representing the feature importance.</p>
</list-item>
<list-item>
<p>Step 2: Aggregate logits from both branches to compute an element- wise selection mask using a sigmoid function. <xref ref-type="disp-formula" rid="e7">Equation 7</xref> defines this aggregation process.</p>
</list-item>
</list>
<disp-formula id="e7">
<mml:math id="m41">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Sigmoid</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>Here, <inline-formula id="inf35">
<mml:math id="m42">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the attention-based selection mask used to balance feature contributions from the two branches. The summed logits <inline-formula id="inf36">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> capture joint channel importance. The sigmoid function constrains the mask values between 0 and 1, enabling soft feature weighting.<list list-type="simple">
<list-item>
<p>Step 3: Compute the final fused output via element-wise multiplication, as specified in <xref ref-type="disp-formula" rid="e8">Equation 8</xref>:</p>
</list-item>
</list>
<disp-formula id="e8">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>output</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2297;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2297;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>Here, <inline-formula id="inf37">
<mml:math id="m45">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf38">
<mml:math id="m46">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represent the output features from the Transformer and convolutional branches, respectively. <inline-formula id="inf39">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>output</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the final fused representation. The selection mask <inline-formula id="inf40">
<mml:math id="m48">
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> adaptively controls the contribution of each branch, enabling dynamic integration of global and local information.</p>
</sec>
<sec id="s2-4">
<title>2.4 Loss function</title>
<p>To supervise both the coarse and fine segmentation branches during training, a Progressive Dual-Branch Loss (PDB Loss) is proposed. This loss function dynamically adjusts the supervision weights between the coarse and fine predictions over training epochs. The total training loss is precisely defined by <xref ref-type="disp-formula" rid="e9">Equation 9</xref>:<disp-formula id="e9">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>PDB</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>DiceCE</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mtext>coarse</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>DiceCE</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mtext>fine</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>Here, <inline-formula id="inf41">
<mml:math id="m50">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mtext>coarse</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf42">
<mml:math id="m51">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mtext>fine</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> are the predicted masks from the coarse and fine branches for the <inline-formula id="inf43">
<mml:math id="m52">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th sample, and <inline-formula id="inf44">
<mml:math id="m53">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the corresponding ground truth. <inline-formula id="inf45">
<mml:math id="m54">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the batch size. <inline-formula id="inf46">
<mml:math id="m55">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is a progressive weight that determines the relative contribution of the fine branch.</p>
<p>For each prediction, a hybrid loss combining Dice and binary cross&#x2011;entropy (BCE) is used, aspresented in <xref ref-type="disp-formula" rid="e10">Equation 10</xref>:<disp-formula id="e10">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>DiceCE</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>dice</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>Dice</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>ce</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>CE</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>The loss weights were set as <inline-formula id="inf47">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">dice</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.8</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf48">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. To shift the learning focus from coarse to fine predictions over time, the coefficient <inline-formula id="inf49">
<mml:math id="m59">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> was scheduled according to the current epoch <inline-formula id="inf50">
<mml:math id="m60">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as given in <xref ref-type="disp-formula" rid="e11">Equation 11</xref>:<disp-formula id="e11">
<mml:math id="m61">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>min</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mn>1.0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
</p>
<p>This progressive weighting strategy encourages the model to learn global structural features in early epochs via the coarse branch and gradually refine local boundaries and details through the fine branch.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Experiments</title>
<p>In this section, a series of comprehensive experiments is performed to evaluate the effectiveness of the proposed HiImp-SMI on medical image segmentation tasks. Initially, the experimental setup is detailed, including dataset selection and training configurations. Subsequently, the performance of HiImp-SMI is quantitatively and qualitatively compared with state-of-the-art implicit and discrete segmentation approaches, specifically addressing binary polyp segmentation on the Kvasir-Sessile dataset [<xref ref-type="bibr" rid="B13">13</xref>] and multi-class organ segmentation on the BCV dataset [<xref ref-type="bibr" rid="B56">56</xref>]. Additionally, robustness analyses under various data distributions are presented. Finally, a systematic ablation study is conducted to elucidate the contributions of individual modules within HiImp-SMI.</p>
<p>The quantitative comparison results are summarized in <xref ref-type="table" rid="T1">Table 1</xref>, highlighting mean Dice and IoU scores alongside corresponding standard deviations. The best-performing methods are emphasized in bold, illustrating that HiImp-SMI consistently achieves superior segmentation performance compared to existing state-of-the-art methods.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Overall segmentation results compared to state-of-the-art discrete and implicit methods. The last two columns present the mean Dice and IoU scores with standard deviation. The best results are highlighted in bold.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Method type</th>
<th rowspan="2" align="left">Method</th>
<th colspan="2" align="center">Kvasir-sessile</th>
<th colspan="2" align="center">BCV</th>
</tr>
<tr>
<th align="center">Dice(%)</th>
<th align="center">IoU(%)</th>
<th align="center">Dice(%)</th>
<th align="center">IoU(%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="6" align="left">Discrete</td>
<td align="left">U-Net [<xref ref-type="bibr" rid="B1">1</xref>]</td>
<td align="center">63.89<inline-formula id="inf51">
<mml:math id="m62">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>1.30</td>
<td align="center">46.94<inline-formula id="inf52">
<mml:math id="m63">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.65</td>
<td align="center">74.47<inline-formula id="inf53">
<mml:math id="m64">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>1.57</td>
<td align="center">59.32<inline-formula id="inf54">
<mml:math id="m65">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.79</td>
</tr>
<tr>
<td align="left">PraNet [<xref ref-type="bibr" rid="B3">3</xref>]</td>
<td align="center">82.56<inline-formula id="inf55">
<mml:math id="m66">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>1.08</td>
<td align="center">70.3<inline-formula id="inf56">
<mml:math id="m67">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.54</td>
<td align="center">N/A</td>
<td align="center">N/A</td>
</tr>
<tr>
<td align="left">UNETR [<xref ref-type="bibr" rid="B11">11</xref>]</td>
<td align="center">N/A</td>
<td align="center">N/A</td>
<td align="center">81.14<inline-formula id="inf57">
<mml:math id="m68">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.85</td>
<td align="center">68.27<inline-formula id="inf58">
<mml:math id="m69">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.43</td>
</tr>
<tr>
<td align="left">Res2UNet [<xref ref-type="bibr" rid="B9">9</xref>]</td>
<td align="center">81.62<inline-formula id="inf59">
<mml:math id="m70">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.97</td>
<td align="center">68.95<inline-formula id="inf60">
<mml:math id="m71">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.49</td>
<td align="center">79.23<inline-formula id="inf61">
<mml:math id="m72">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.66</td>
<td align="center">65.6<inline-formula id="inf62">
<mml:math id="m73">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.33</td>
</tr>
<tr>
<td align="left">NnUNet [<xref ref-type="bibr" rid="B2">2</xref>]</td>
<td align="center">82.97<inline-formula id="inf63">
<mml:math id="m74">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.89</td>
<td align="center">70.9<inline-formula id="inf64">
<mml:math id="m75">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.45</td>
<td align="center">85.15<inline-formula id="inf65">
<mml:math id="m76">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.67</td>
<td align="center">74.14<inline-formula id="inf66">
<mml:math id="m77">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.34</td>
</tr>
<tr>
<td align="left">MedSAM [<xref ref-type="bibr" rid="B15">15</xref>]</td>
<td align="center">82.88<inline-formula id="inf67">
<mml:math id="m78">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.55</td>
<td align="center">70.77<inline-formula id="inf68">
<mml:math id="m79">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.28</td>
<td align="center">85.85<inline-formula id="inf69">
<mml:math id="m80">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.81</td>
<td align="center">75.21<inline-formula id="inf70">
<mml:math id="m81">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.41</td>
</tr>
<tr>
<td rowspan="5" align="left">Implicit</td>
<td align="left">OSSNet [<xref ref-type="bibr" rid="B44">44</xref>]</td>
<td align="center">76.11<inline-formula id="inf71">
<mml:math id="m82">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>1.14</td>
<td align="center">61.43<inline-formula id="inf72">
<mml:math id="m83">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.57</td>
<td align="center">73.38<inline-formula id="inf73">
<mml:math id="m84">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>1.65</td>
<td align="center">57.95<inline-formula id="inf74">
<mml:math id="m85">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.83</td>
</tr>
<tr>
<td align="left">IOSNet [<xref ref-type="bibr" rid="B45">45</xref>]</td>
<td align="center">78.37<inline-formula id="inf75">
<mml:math id="m86">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.76</td>
<td align="center">64.43<inline-formula id="inf76">
<mml:math id="m87">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.38</td>
<td align="center">76.75<inline-formula id="inf77">
<mml:math id="m88">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>1.37</td>
<td align="center">62.27<inline-formula id="inf78">
<mml:math id="m89">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.69</td>
</tr>
<tr>
<td align="left">SWIPE [<xref ref-type="bibr" rid="B46">46</xref>]</td>
<td align="center">85.05<inline-formula id="inf79">
<mml:math id="m90">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.82</td>
<td align="center">73.99<inline-formula id="inf80">
<mml:math id="m91">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.41</td>
<td align="center">81.21<inline-formula id="inf81">
<mml:math id="m92">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.94</td>
<td align="center">68.36<inline-formula id="inf82">
<mml:math id="m93">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.47</td>
</tr>
<tr>
<td align="left">I-MedSAM [<xref ref-type="bibr" rid="B55">55</xref>]</td>
<td align="center">91.49<inline-formula id="inf83">
<mml:math id="m94">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.52</td>
<td align="center">84.31<inline-formula id="inf84">
<mml:math id="m95">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.26</td>
<td align="center">89.91<inline-formula id="inf85">
<mml:math id="m96">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.68</td>
<td align="center">81.67<inline-formula id="inf86">
<mml:math id="m97">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>0.34</td>
</tr>
<tr>
<td align="left">HiImp-SMI (Ours)</td>
<td align="center">
<bold>92.39</bold>
<inline-formula id="inf87">
<mml:math id="m98">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>0.36</bold>
</td>
<td align="center">
<bold>85.86</bold>
<inline-formula id="inf88">
<mml:math id="m99">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>0.18</bold>
</td>
<td align="center">
<bold>91.21</bold>
<inline-formula id="inf89">
<mml:math id="m100">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>0.31</bold>
</td>
<td align="center">
<bold>83.84</bold>
<inline-formula id="inf90">
<mml:math id="m101">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>0.16</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note. &#x201c;N/A&#x201d; indicates that the corresponding experiment was not conducted.</p>
</fn>
<fn>
<p>Bold values indicate the best performance for each metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<sec id="s3-1">
<title>3.1 Experimental setup</title>
<p>The model&#x2019;s performance is evaluated on two distinct medical image segmentation tasks: binary polyp segmentation and multi-class abdominal organ segmentation.</p>
<p>For polyp segmentation, experiments are conducted on the challenging Kvasir-Sessile dataset [<xref ref-type="bibr" rid="B13">13</xref>], which contains 196 RGB images of small sessile polyps. To assess the generalization capability of HiImp-SMI, the pretrained model is further evaluated on the CVC-ClinicDB dataset [<xref ref-type="bibr" rid="B13">13</xref>], which consists of 612 images extracted from 31 colonoscopy sequences.</p>
<p>For multi-organ segmentation, the model is trained on the BCV dataset [<xref ref-type="bibr" rid="B56">56</xref>], which includes 30 CT scans with annotations for 13 organs, and is further evaluated on the AMOS dataset [<xref ref-type="bibr" rid="B57">57</xref>], which contains 200 CT training samples, following the same experimental setup as [<xref ref-type="bibr" rid="B22">22</xref>]. Since this study focuses on 2D medical image segmentation, slice-wise segmentation is performed on CT images. Following the data preprocessing strategy of SWIPE [<xref ref-type="bibr" rid="B46">46</xref>], all datasets are split into training, validation, and test sets in a 6:2:2 ratio, and the reported Dice scores are based on test set results.</p>
<p>The training process involves fine-tuning the SAM encoder [<xref ref-type="bibr" rid="B7">7</xref>] with ViT-B as the backbone network. The LoRA rank is set to 4, with amplitude information incorporated in the frequency adapter. The MLP dimensions for the implicit segmentation decoder are [1,024, 512] for Decc and [512, 256, 256, 128] for Decf. During training, 12.5% of the most uncertain points are sampled for refinement, and the dropout probability is set to 0.5. For the multi-organ segmentation task, the final layer of Decc and Decf is adjusted to match the number of target segmentation classes. HiImp-SMI is optimized using AdamW [<xref ref-type="bibr" rid="B58">58</xref>] with <inline-formula id="inf91">
<mml:math id="m102">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, a learning rate of <inline-formula id="inf92">
<mml:math id="m103">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">ada</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for the encoder adapter, and <inline-formula id="inf93">
<mml:math id="m104">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">dec</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for the decoder.</p>
<p>To ensure fair comparison, all methods are trained for 1,000 epochs under the same experimental setup. During testing, Dice scores and Hausdorff distances [<xref ref-type="bibr" rid="B6">6</xref>] are reported based on the best validation epoch. The input image resolutions are set to <inline-formula id="inf94">
<mml:math id="m105">
<mml:mrow>
<mml:mn>384</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>384</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> (Sessile dataset) and <inline-formula id="inf95">
<mml:math id="m106">
<mml:mrow>
<mml:mn>512</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>512</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> (BCV dataset slices).</p>
<p>The baseline approaches are categorized into discrete methods and implicit (continuous) methods. The discrete methods include U-Net [<xref ref-type="bibr" rid="B1">1</xref>], PraNet [<xref ref-type="bibr" rid="B3">3</xref>], Res2UNet [<xref ref-type="bibr" rid="B9">9</xref>], nnUNet [<xref ref-type="bibr" rid="B2">2</xref>], UNETR [<xref ref-type="bibr" rid="B11">11</xref>], and MedSAM [<xref ref-type="bibr" rid="B15">15</xref>]. Among these, MedSAM [<xref ref-type="bibr" rid="B15">15</xref>] is also a SAM-based approach, where the original decoder is directly fine-tuned. The implicit methods include OSSNet [<xref ref-type="bibr" rid="B44">44</xref>], IOSNet [<xref ref-type="bibr" rid="B45">45</xref>], and SWIPE [<xref ref-type="bibr" rid="B46">46</xref>] and I-MedSAM [<xref ref-type="bibr" rid="B55">55</xref>].</p>
</sec>
<sec id="s3-2">
<title>3.2 Quantitative comparison</title>
<p>A Dice score comparison is first presented against baseline methods. Subsequently, experiments are conducted across different resolutions and domains to evaluate the model&#x2019;s cross-domain generalization ability under data distribution shifts. Finally, Hausdorff Distance (HD) [<xref ref-type="bibr" rid="B6">6</xref>] is computed to compare the segmentation boundary quality across different experimental settings.</p>
<p>Discrete methods and implicit methods are compared in terms of trainable parameters and Dice scores (including standard deviation). Specifically, binary segmentation is performed on the Kvasir-Sessile dataset, while multi-class segmentation is conducted on the CT BCV dataset, with results detailed in <xref ref-type="table" rid="T2">Table 2</xref>. Leveraging the proposed frequency adapter, SAM generates richer feature representations, leading to improved segmentation boundary quality. In contrast, SwIPE, which employs Res2Net-50 [<xref ref-type="bibr" rid="B9">9</xref>] as its backbone, exhibits weaker feature extraction capability, resulting in lower segmentation quality.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Cross-resolution evaluation from <inline-formula id="inf96">
<mml:math id="m107">
<mml:mrow>
<mml:mn>384</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>384</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf97">
<mml:math id="m108">
<mml:mrow>
<mml:mn>128</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>128</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and from <inline-formula id="inf98">
<mml:math id="m109">
<mml:mrow>
<mml:mn>384</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>384</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf99">
<mml:math id="m110">
<mml:mrow>
<mml:mn>896</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>896</mml:mn>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Method type</th>
<th rowspan="2" align="left">Method</th>
<th colspan="2" align="center">
<bold>384</bold>
<inline-formula id="inf100">
<mml:math id="m111">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>384</bold>
<inline-formula id="inf101">
<mml:math id="m112">
<mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>128</bold>
<inline-formula id="inf102">
<mml:math id="m113">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>128</bold>
</th>
<th colspan="2" align="center">
<bold>384</bold>
<inline-formula id="inf103">
<mml:math id="m114">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>384</bold>
<inline-formula id="inf104">
<mml:math id="m115">
<mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>896</bold>
<inline-formula id="inf105">
<mml:math id="m116">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>896</bold>
</th>
</tr>
<tr>
<th align="center">Dice(%)</th>
<th align="center">IoU(%)</th>
<th align="center">Dice(%)</th>
<th align="center">IoU(%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="5" align="left">Discrete</td>
<td align="left">PraNet [<xref ref-type="bibr" rid="B3">3</xref>]</td>
<td align="center">72.64</td>
<td align="center">57.04</td>
<td align="center">74.95</td>
<td align="center">59.94</td>
</tr>
<tr>
<td align="left">PraNet&#x2a; [<xref ref-type="bibr" rid="B3">3</xref>]</td>
<td align="center">68.79</td>
<td align="center">52.43</td>
<td align="center">43.92</td>
<td align="center">28.14</td>
</tr>
<tr>
<td align="left">nnUNet [<xref ref-type="bibr" rid="B2">2</xref>]</td>
<td align="center">73.97</td>
<td align="center">58.69</td>
<td align="center">83.56</td>
<td align="center">71.76</td>
</tr>
<tr>
<td align="left">nnUNet&#x2a; [<xref ref-type="bibr" rid="B2">2</xref>]</td>
<td align="center">65.34</td>
<td align="center">48.52</td>
<td align="center">76.36</td>
<td align="center">61.76</td>
</tr>
<tr>
<td align="left">MedSAM [<xref ref-type="bibr" rid="B15">15</xref>]</td>
<td align="center">82.39</td>
<td align="center">70.05</td>
<td align="center">83.56</td>
<td align="center">71.76</td>
</tr>
<tr>
<td rowspan="4" align="left">Implicit</td>
<td align="left">IOSNet [<xref ref-type="bibr" rid="B45">45</xref>]</td>
<td align="center">78.37</td>
<td align="center">64.43</td>
<td align="center">78.01</td>
<td align="center">63.95</td>
</tr>
<tr>
<td align="left">SWIPE [<xref ref-type="bibr" rid="B46">46</xref>]</td>
<td align="center">81.26</td>
<td align="center">68.44</td>
<td align="center">84.33</td>
<td align="center">72.91</td>
</tr>
<tr>
<td align="left">I-MedSAM [<xref ref-type="bibr" rid="B55">55</xref>]</td>
<td align="center">91.45</td>
<td align="center">84.25</td>
<td align="center">91.33</td>
<td align="center">84.04</td>
</tr>
<tr>
<td align="left">HiImp-SMI (ours)</td>
<td align="center">
<bold>92.52</bold>
</td>
<td align="center">
<bold>86.08</bold>
</td>
<td align="center">
<bold>92.28</bold>
</td>
<td align="center">
<bold>85.67</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best performance for each metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The adaptability of binary polyp segmentation across different resolutions and domains is assessed by comparing it with the best-performing discrete and implicit methods. To adapt to different target resolutions (e.g., low resolution <inline-formula id="inf106">
<mml:math id="m117">
<mml:mrow>
<mml:mn>128</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>128</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and high resolution <inline-formula id="inf107">
<mml:math id="m118">
<mml:mrow>
<mml:mn>896</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>896</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>), the pretrained HiImp-SMI model, initially trained at <inline-formula id="inf108">
<mml:math id="m119">
<mml:mrow>
<mml:mn>384</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>384</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> standard resolution, is modified by scaling the input coordinates to match the target resolution, and the corresponding Dice scores are computed. For discrete methods, the output resolution remains consistent with the input resolution. Input images at the original resolution of <inline-formula id="inf109">
<mml:math id="m120">
<mml:mrow>
<mml:mn>384</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>384</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> are provided, and the generated segmentation results are rescaled to the target resolution for evaluation. Additionally, the suffix (&#x2a;) is used to mark discrete baselines, where the original medical images are resized to the target resolution before being fed into the models, allowing these methods to directly generate segmentation results at the target resolution.</p>
<p>As shown in <xref ref-type="table" rid="T2">Table 2</xref>, implicit methods exhibit stronger adaptability to spatial resolution changes and consistently outperform discrete methods. Among implicit methods, HiImp-SMI achieves the highest performance across different output resolutions, which can be attributed to the proposed frequency adapter, enhancing HiImp-SMI&#x2019;s predictive capability across resolutions.</p>
<p>Model performance across different datasets is examined. In binary polyp segmentation, all methods are pretrained on the Kvasir-Sessile dataset and directly evaluated on the CVC dataset. Similarly, in multi-class abdominal organ segmentation, all methods are pretrained on the BCV dataset and evaluated on the AMOS dataset, focusing exclusively on the liver class.</p>
<p>As shown in <xref ref-type="table" rid="T3">Table 3</xref>, leveraging SAM&#x2019;s generalization ability, HiImp-SMI outperforms the best discrete method, achieving Dice scores of 91.58% on the CVC dataset and 88.17% on the AMOS dataset.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Cross-domain results for binary polyp segmentation and multi-class abdominal organ segmentation.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">Method type</th>
<th rowspan="2" align="left">Method</th>
<th colspan="2" align="center">Kvasir<inline-formula id="inf110">
<mml:math id="m121">
<mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>CVC</th>
<th colspan="2" align="center">BCV<inline-formula id="inf111">
<mml:math id="m122">
<mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>AMOS</th>
</tr>
<tr>
<th align="center">Dice(%)</th>
<th align="center">IoU(%)</th>
<th align="center">Dice(%)</th>
<th align="center">IoU(%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="left">Discrete</td>
<td align="left">PraNet [<xref ref-type="bibr" rid="B3">3</xref>]</td>
<td align="center">68.37</td>
<td align="center">51.94</td>
<td align="center">N/A</td>
<td align="center">N/A</td>
</tr>
<tr>
<td align="left">UNETR [<xref ref-type="bibr" rid="B11">11</xref>]</td>
<td align="center">N/A</td>
<td align="center">N/A</td>
<td align="center">81.75</td>
<td align="center">69.13</td>
</tr>
<tr>
<td align="left">nnUNet [<xref ref-type="bibr" rid="B2">2</xref>]</td>
<td align="center">84.91</td>
<td align="center">73.78</td>
<td align="center">79.63</td>
<td align="center">66.15</td>
</tr>
<tr>
<td align="left">MedSAM [<xref ref-type="bibr" rid="B15">15</xref>]</td>
<td align="center">74.59</td>
<td align="center">59.48</td>
<td align="center">71.98</td>
<td align="center">56.23</td>
</tr>
<tr>
<td rowspan="4" align="left">Implicit</td>
<td align="left">IOSNet [<xref ref-type="bibr" rid="B45">45</xref>]</td>
<td align="center">59.42</td>
<td align="center">42.27</td>
<td align="center">79.48</td>
<td align="center">65.95</td>
</tr>
<tr>
<td align="left">SWIPE [<xref ref-type="bibr" rid="B46">46</xref>]</td>
<td align="center">70.1</td>
<td align="center">53.96</td>
<td align="center">82.81</td>
<td align="center">70.66</td>
</tr>
<tr>
<td align="left">I-MedSAM [<xref ref-type="bibr" rid="B55">55</xref>]</td>
<td align="center">88.83</td>
<td align="center">79.9</td>
<td align="center">86.28</td>
<td align="center">75.87</td>
</tr>
<tr>
<td align="left">HiImp-SMI (ours)</td>
<td align="center">
<bold>91.58</bold>
</td>
<td align="center">
<bold>84.47</bold>
</td>
<td align="center">
<bold>88.17</bold>
</td>
<td align="center">
<bold>78.84</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note.&#x201c;N/A&#x2033; indicates that the corresponding experiment was not conducted.</p>
</fn>
<fn>
<p>Bold values indicate the best performance for each metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Segmentation boundary quality is further assessed using Hausdorff Distance (HD) [<xref ref-type="bibr" rid="B19">19</xref>]. As shown in <xref ref-type="table" rid="T4">Table 4</xref>, HiImp-SMI achieves lower HD scores, indicating superior boundary precision compared to existing methods.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>HD distance <inline-formula id="inf112">
<mml:math id="m123">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x2193;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> for different methods and datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="center">Kvasir-sessile</th>
<th align="center">Kvasir<inline-formula id="inf113">
<mml:math id="m124">
<mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>CVC</th>
<th align="center">
<bold>384</bold>
<inline-formula id="inf114">
<mml:math id="m125">
<mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>128</bold>
</th>
<th align="center">
<bold>384</bold>
<inline-formula id="inf115">
<mml:math id="m126">
<mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
<bold>896</bold>
</th>
<th align="center">BCV</th>
<th align="center">BCV<inline-formula id="inf116">
<mml:math id="m127">
<mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>AMOS</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">nnUNet [<xref ref-type="bibr" rid="B2">2</xref>]</td>
<td align="center">31.30</td>
<td align="center">82.31</td>
<td align="center">13.69</td>
<td align="center">72.31</td>
<td align="center">6.50</td>
<td align="center">80.39</td>
</tr>
<tr>
<td align="left">MedSAM [<xref ref-type="bibr" rid="B15">15</xref>]</td>
<td align="center">21.53</td>
<td align="center">30.15</td>
<td align="center">8.04</td>
<td align="center">51.82</td>
<td align="center">10.62</td>
<td align="center">52.14</td>
</tr>
<tr>
<td align="left">IOSNet [<xref ref-type="bibr" rid="B45">45</xref>]</td>
<td align="center">51.72</td>
<td align="center">81.60</td>
<td align="center">35.33</td>
<td align="center">87.86</td>
<td align="center">21.46</td>
<td align="center">61.19</td>
</tr>
<tr>
<td align="left">I-MedSAM [<xref ref-type="bibr" rid="B55">55</xref>]</td>
<td align="center">11.59</td>
<td align="center">
<bold>19.76</bold>
</td>
<td align="center">7.91</td>
<td align="center">32.77</td>
<td align="center">5.95</td>
<td align="center">
<bold>37.53</bold>
</td>
</tr>
<tr>
<td align="left">HiImp-SMI (ours)</td>
<td align="center">
<bold>10.48</bold>
</td>
<td align="center">20.30</td>
<td align="center">
<bold>3.60</bold>
</td>
<td align="center">
<bold>24.52</bold>
</td>
<td align="center">
<bold>4.97</bold>
</td>
<td align="center">38.12</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best performance for each metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-3">
<title>3.3 Qualitative comparison</title>
<p>As shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, a qualitative comparison is conducted on the Kvasir-Sessile dataset. Additionally, the input medical images and their corresponding ground truth segmentation masks are provided, where segmentation boundaries are highlighted in green in <xref ref-type="fig" rid="F5">Figure 5</xref>. The sharpness of boundaries in the visual results may be attributed in part to the frequency-domain information introduced via FFT.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Qualitative comparisons on five representative samples. The last row indicates the method names corresponding to each column.</p>
</caption>
<graphic xlink:href="fphy-13-1614983-g005.tif">
<alt-text content-type="machine-generated">A grid of images showing colonoscopy results for segmentation methods comparison, including Ground Truth, IOSNet, nnUNet, MedSAM, I-MedSAM, and HiImp-SMI, with various highlighted areas for each approach.</alt-text>
</graphic>
</fig>
<p>From the results, it is evident that HiImp-SMI produces more precise segmentation boundaries. By leveraging the proposed modules, HiImp-SMI effectively aggregates high-frequency information from the input, leading to improved segmentation accuracy in the final output.</p>
</sec>
<sec id="s3-4">
<title>3.4 Ablation study</title>
<p>An ablation study is conducted to evaluate the effectiveness of each module within the high-frequency adapter. The results are summarized in <xref ref-type="table" rid="T5">Table 5</xref>.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Ablation study on the integration of different modules: Channel Attention Block (CAB), Multi-branch Cross Attention Block (MCAB), and ViT-Conv Fusion Block (VCFB). Evaluation is conducted on the Kvasir-Sessile dataset and its cross-domain transfer to the CVC dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="3" align="center">Modules</th>
<th colspan="3" align="center">Kvasir-sessile</th>
<th colspan="3" align="center">Kvasir-sessile <inline-formula id="inf117">
<mml:math id="m128">
<mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> CVC</th>
</tr>
<tr>
<th align="center">CAB</th>
<th align="center">MCAB</th>
<th align="center">VCFB</th>
<th align="center">Dice (%) <inline-formula id="inf118">
<mml:math id="m129">
<mml:mrow>
<mml:mi>&#x2191;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">HD <inline-formula id="inf119">
<mml:math id="m130">
<mml:mrow>
<mml:mi>&#x2193;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">IoU (%) <inline-formula id="inf120">
<mml:math id="m131">
<mml:mrow>
<mml:mi>&#x2191;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">Dice (%) <inline-formula id="inf121">
<mml:math id="m132">
<mml:mrow>
<mml:mi>&#x2191;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">HD <inline-formula id="inf122">
<mml:math id="m133">
<mml:mrow>
<mml:mi>&#x2193;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">IoU (%) <inline-formula id="inf123">
<mml:math id="m134">
<mml:mrow>
<mml:mi>&#x2191;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">91.81</td>
<td align="center">11.80</td>
<td align="center">84.86</td>
<td align="center">89.07</td>
<td align="center">24.06</td>
<td align="center">80.29</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf124">
<mml:math id="m135">
<mml:mrow>
<mml:mi>&#x2713;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left"/>
<td align="left"/>
<td align="center">92.02</td>
<td align="center">11.28</td>
<td align="center">85.22</td>
<td align="center">88.94</td>
<td align="center">24.66</td>
<td align="center">80.08</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf125">
<mml:math id="m136">
<mml:mrow>
<mml:mi>&#x2713;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf126">
<mml:math id="m137">
<mml:mrow>
<mml:mi>&#x2713;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left"/>
<td align="center">92.42</td>
<td align="center">11.50</td>
<td align="center">85.91</td>
<td align="center">88.87</td>
<td align="center">22.12</td>
<td align="center">79.97</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf127">
<mml:math id="m138">
<mml:mrow>
<mml:mi>&#x2713;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf128">
<mml:math id="m139">
<mml:mrow>
<mml:mi>&#x2713;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf129">
<mml:math id="m140">
<mml:mrow>
<mml:mi>&#x2713;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<bold>92.51</bold>
</td>
<td align="center">
<bold>9.98</bold>
</td>
<td align="center">
<bold>86.06</bold>
</td>
<td align="center">
<bold>91.46</bold>
</td>
<td align="center">
<bold>21.03</bold>
</td>
<td align="center">
<bold>84.26</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the best performance for each metric.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In the baseline model, the single frequency adapter module consists of a linear down-projection layer, a GELU activation function, and a linear up-projection layer. On the Kvasir-Sessile dataset [<xref ref-type="bibr" rid="B8">8</xref>], the baseline model achieves a Dice score of 91.81% and an HD of 11.80. When transferred to the CVC dataset, the Dice score drops to 89.07%, with an HD of 24.06.</p>
<p>As the channel attention block, bi-directional cross-attention block, and ViT-Conv fusion block are incrementally added, model performance exhibits a significant improvement. When all three modules are incorporated, the Dice score on the Kvasir-Sessile dataset improves to 92.51%, while HD decreases to 9.98. Similarly, on the CVC dataset, the Dice score improves to 91.46%, and HD decreases to 21.03, highlighting the necessity and effectiveness of the proposed modules.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>In this study, a novel implicit Transformer-based framework, HiImp-SMI, was proposed to overcome key limitations in medical image segmentation, such as poor boundary refinement, weak feature fusion, and limited cross-domain generalization. High-frequency information and multi-scale features were incorporated through three main components: a Channel Attention Block for frequency-domain feature adaptation, a Multi-Branch Cross Attention Block for hierarchical feature exchange, and a ViT-Conv Fusion Block for adaptive context integration. Additionally, a Progressive Dual-Branch Loss was introduced to guide the training process from coarse to fine segmentation. Extensive experiments conducted on the Kvasir-Sessile and BCV datasets demonstrated that HiImp-SMI consistently outperformed state-of-the-art methods, particularly in cross-domain and cross-resolution tasks. Ablation studies further confirmed the effectiveness of each proposed module.</p>
<p>However, the current framework has not yet been validated in clinical or multi-center settings. Future research will aim to evaluate its applicability in real-world clinical workflows.</p>
<p>Overall, HiImp-SMI provided a unified and adaptive solution for precise and generalizable medical image segmentation.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: Kvasir-Sessile Dataset: <ext-link ext-link-type="uri" xlink:href="https://datasets.simula.no/kvasir/">https://datasets.simula.no/kvasir/</ext-link> Repository: Simula Research Laboratory Accession Number: Not applicable (open access dataset) CVC-ClinicDB: <ext-link ext-link-type="uri" xlink:href="https://github.com/CVC-ClinicDB">https://github.com/CVC-ClinicDB</ext-link> Repository: GitHub Accession Number: Not applicableBCV (Beyond Cranial Vault) Dataset: <ext-link ext-link-type="uri" xlink:href="https://www.synapse.org/#!Synapse:syn3193805">https://www.synapse.org/&#x23;!Synapse:syn3193805</ext-link> Repository: Synapse Accession Number: syn3193805AMOS Dataset: <ext-link ext-link-type="uri" xlink:href="https://amos22.grand-challenge.org/">https://amos22.grand-challenge.org/</ext-link> Repository: Grand Challenge Accession Number: Not applicable.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>LH: Conceptualization, Methodology, Validation, Writing &#x2013; original draft, Supervision, Writing &#x2013; review and editing, Data curation. FP: Supervision, Writing &#x2013; review and editing. BH: Data curation, Writing &#x2013; review and editing, Investigation. YC: Supervision, Writing &#x2013; review and editing, Methodology, Project administration.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the Liaoning Provincial Science and Technology Plan Joint Project (Grant No. 2024-MSLH-033).</p>
</sec>
<ack>
<p>Many thanks to Yinghong Cao and Feng Peng for their help in achieving this work.</p>
</ack>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ronneberger</surname>
<given-names>O</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Brox</surname>
<given-names>T</given-names>
</name>
</person-group>. <article-title>U-net: convolutional networks for biomedical image segmentation</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Navab</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Hornegger</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Wells</surname>
<given-names>WM</given-names>
</name>
<name>
<surname>Frangi</surname>
<given-names>AF</given-names>
</name>
</person-group>, editors. <conf-name>Proceedings of the 18th International Conference on Medical Image Computing and Computer-Assisted Intervention (MICCAI 2015)</conf-name>, <volume>9351</volume>. <publisher-loc>Cham, Switzerland: Springer</publisher-loc> (<year>2015</year>). p. <fpage>234</fpage>&#x2013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-24574-4_28</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Isensee</surname>
<given-names>F</given-names>
</name>
<name>
<surname>J&#xe4;ger</surname>
<given-names>PF</given-names>
</name>
<name>
<surname>Kohl</surname>
<given-names>SAA</given-names>
</name>
<name>
<surname>Petersen</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Maier-Hein</surname>
<given-names>KH</given-names>
</name>
</person-group>. <article-title>Automated design of deep learning methods for biomedical image segmentation</article-title>. <source>arXiv preprint arXiv:1904</source> (<year>2019</year>):<fpage>08128</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1904.08128</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Fan</surname>
<given-names>DP</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>GP</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>J</given-names>
</name>
<etal/>
</person-group> <article-title>Pranet: parallel reverse attention network for polyp segmentation</article-title>. In: <conf-name>Proceedings of the 23rd International Conference on Medical Image Computing and Computer-Assisted Intervention (MICCAI 2020)</conf-name>, <volume>12266</volume>. <publisher-loc>Cham, Switzerland: Springer</publisher-loc> (<year>2020</year>). p. <fpage>263</fpage>&#x2013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-59725-2_26</pub-id>
<source>Lecture Notes in Computer Sci</source>
</citation>
</ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alahmadi</surname>
<given-names>MD</given-names>
</name>
</person-group>. <article-title>Boundary aware u-net for medical image segmentation</article-title>. <source>Arabian J Sci Eng</source> (<year>2023</year>) <volume>48</volume>:<fpage>9929</fpage>&#x2013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1007/s13369-022-07431-y</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bernal</surname>
<given-names>J</given-names>
</name>
<name>
<surname>S&#xe1;nchez</surname>
<given-names>FJ</given-names>
</name>
<name>
<surname>Fern&#xe1;ndez-Esparrach</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Gil</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Rodr&#xed;guez</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Vilari&#xf1;o</surname>
<given-names>F</given-names>
</name>
</person-group>. <article-title>Wm-dova maps for accurate polyp highlighting in colonoscopy: validation vs. saliency maps from physicians</article-title>. <source>Comput Med Imaging Graphics</source> (<year>2015</year>) <volume>43</volume>:<fpage>99</fpage>&#x2013;<lpage>111</lpage>. <pub-id pub-id-type="doi">10.1016/j.compmedimag.2015.02.007</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huttenlocher</surname>
<given-names>DP</given-names>
</name>
<name>
<surname>Klanderman</surname>
<given-names>GA</given-names>
</name>
<name>
<surname>Rucklidge</surname>
<given-names>WJ</given-names>
</name>
</person-group>. <article-title>Comparing images using the hausdorff distance</article-title>. <source>IEEE Trans Pattern Anal Machine Intelligence</source> (<year>1993</year>) <volume>15</volume>:<fpage>850</fpage>&#x2013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.1109/34.232073</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gal</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Ghahramani</surname>
<given-names>Z</given-names>
</name>
</person-group>. <article-title>Dropout as a bayesian approximation: representing model uncertainty in deep learning</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Balcan</surname>
<given-names>MF</given-names>
</name>
<name>
<surname>Weinberger</surname>
<given-names>KQ</given-names>
</name>
</person-group>, editors. <source>Proceedings of the 33rd international conference on machine learning</source>, <publisher-loc>New York, NY, USA: PMLR</publisher-loc> (<year>2016</year>). p. <volume>48</volume>. <fpage>1050</fpage>&#x2013;<lpage>9</lpage>.<source>Proc Machine Learn Res</source>
</citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Pleiss</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Weinberger</surname>
<given-names>KQ</given-names>
</name>
</person-group>. <article-title>On calibration of modern neural networks</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Precup</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Teh</surname>
<given-names>YW</given-names>
</name>
</person-group>, editors. <source>Proceedings of the 34th international conference on machine learning</source>, <publisher-loc>Sydney, Australia: PMLR</publisher-loc> (<year>2017</year>). p. <fpage>1321</fpage>&#x2013;<lpage>30</lpage>.<source>Proc Machine Learn Res</source>
</citation>
</ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>SH</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>MM</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>XY</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>MH</given-names>
</name>
<name>
<surname>Torr</surname>
<given-names>P</given-names>
</name>
</person-group>. <article-title>Res2net: a new multi-scale backbone architecture</article-title>. <source>IEEE Trans Pattern Anal Machine Intelligence</source> (<year>2021</year>) <volume>43</volume>:<fpage>652</fpage>&#x2013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2019.2938758</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Mei</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Q</given-names>
</name>
<etal/>
</person-group> <article-title>Transunet: rethinking the u-net architecture design for medical image segmentation through the lens of transformers</article-title>. <source>Med Image Anal</source> (<year>2024</year>) <volume>84</volume>:<fpage>103280</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2024.103280</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hatamizadeh</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Nath</surname>
<given-names>V</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Myronenko</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Landman</surname>
<given-names>B</given-names>
</name>
<etal/>
</person-group> <article-title>UNETR: transformers for 3D medical image segmentation</article-title>. In: <source>
<italic>Proceedings of the IEEE/CVF winter Conference on Applications of computer vision (WACV)</italic> (waikoloa, HI, USA: ieee)</source> (<year>2022</year>). p. <fpage>574</fpage>&#x2013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1109/WACV51458.2022.00181</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>AN</given-names>
</name>
<etal/>
</person-group> <article-title>Attention is all you need</article-title>. <source>Adv in Neural Inf Process Syst</source> (<year>2017</year>). p. <fpage>5998</fpage>&#x2013;<lpage>6008</lpage>.</citation>
</ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>EJ</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Wallis</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Allen-Zhu</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S</given-names>
</name>
<etal/>
</person-group> <article-title>LoRA: low-rank adaptation of large language models</article-title>. In: <source>Proceedings of the international conference on learning representations (ICLR)</source>. <publisher-loc>Virtual Event</publisher-loc>: <publisher-name>OpenReview.net</publisher-name>
<comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://openreview.net/forum?id=nZeVKeeFYf9">https://openreview.net/forum?id&#x3d;nZeVKeeFYf9</ext-link>
</comment> (<year>2022</year>).</citation>
</ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kirillov</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Mintun</surname>
<given-names>E</given-names>
</name>
<name>
<surname>Ravi</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Rolland</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Gustafson</surname>
<given-names>L</given-names>
</name>
<etal/>
</person-group> <article-title>Segment anything</article-title>. In: <source>
<italic>Proceedings of the IEEE/CVF international Conference on computer vision (ICCV)</italic> (paris, France: ieee)</source> (<year>2023</year>). p. <fpage>4015</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV51070.2023.00371</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>J</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>L</given-names>
</name>
<name>
<surname>You</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>B</given-names>
</name>
</person-group>. <article-title>Segment anything in medical images</article-title>. <source>Nat Commun</source> (<year>2024</year>) <volume>15</volume>:<fpage>654</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-024-44824-z</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McGinnis</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Shit</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>HB</given-names>
</name>
<name>
<surname>Sideri-Lampretsa</surname>
<given-names>V</given-names>
</name>
<name>
<surname>Graf</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Dannecker</surname>
<given-names>M</given-names>
</name>
<etal/>
</person-group> <article-title>Single-subject multi-contrast MRI super-resolution via implicit neural representations</article-title>, <source>Med Image Comput Computer Assisted Intervention &#x2013; MICCAI 2023</source>. (<year>2023</year>). <volume>14230</volume>. <fpage>173</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-43993-3_17</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y</given-names>
</name>
<etal/>
</person-group> <article-title>Segment anything model for medical image analysis: an experimental study</article-title>. <source>Med Image Anal</source> (<year>2023</year>) <volume>89</volume>:<fpage>102918</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2023.102918</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y</given-names>
</name>
<etal/>
</person-group> <article-title>NTO3D: neural target object 3d reconstruction with segment anything</article-title>. In: <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition (CVPR)</source>. <publisher-loc>Seattle, WA, USA: IEEE</publisher-loc> (<year>2024</year>). p. <fpage>20352</fpage>&#x2013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52733.2024.01924</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>D</given-names>
</name>
</person-group>. <article-title>Customized segment anything model for medical image segmentation</article-title>. <comment>
<italic>arXiv preprint arXiv:2304.13785</italic>
</comment> (<year>2023</year>).</citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H</given-names>
</name>
<etal/>
</person-group> <article-title>SAM-Med2D</article-title>. <source>arXiv preprint arXiv:2308.16184</source> (<year>2023</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2308.16184</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bui</surname>
<given-names>NT</given-names>
</name>
<name>
<surname>Hoang</surname>
<given-names>DH</given-names>
</name>
<name>
<surname>Tran</surname>
<given-names>MT</given-names>
</name>
<name>
<surname>Doretto</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Adjeroh</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>B</given-names>
</name>
<etal/>
</person-group> <article-title>SAM3D: segment anything model in volumetric medical images</article-title>. <source>arXiv preprint arXiv:2309.03493</source> (<year>2023</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2309.03493</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Sapkota</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>DZ</given-names>
</name>
</person-group>. <article-title>Keep your friends close and enemies farther: debiasing contrastive learning with spatial priors in 3d radiology images</article-title>. In: <conf-name>Proceedings of the 2022 IEEE international Conference on Bioinformatics and biomedicine (BIBM)</conf-name> <publisher-loc>Las Vegas, NV, USA</publisher-loc>: (<publisher-name>IEEE</publisher-name>) (<year>2022</year>). p. <fpage>1824</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/BIBM55620.2022.9995481</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>X</given-names>
</name>
<etal/>
</person-group> <article-title>Mask-enhanced segment anything model for tumor lesion semantic segmentation</article-title>. In: <source>Proceedings of the international conference on medical image computing and computer-assisted intervention (MICCAI 2024)</source> (<year>2024</year>). p. <fpage>403</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-72111-3_38</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Jahanshahi</surname>
<given-names>H</given-names>
</name>
</person-group>. <article-title>Modeling and analysis of cellular neural networks based on memcapacitor</article-title>. <source>Int J Bifurcation Chaos</source> (<year>2025</year>) <volume>35</volume>. <pub-id pub-id-type="doi">10.1142/S0218127425300101</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>Dynamical analysis, multi-cavity control and dsp implementation of a novel memristive autapse neuron model emulating brain behaviors</article-title>. <source>Chaos, Solitons and Fractals</source> (<year>2025</year>) <volume>191</volume>:<fpage>115857</fpage>. <pub-id pub-id-type="doi">10.1016/j.chaos.2024.115857</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Jahanshahi</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Alkhateeb</surname>
<given-names>AF</given-names>
</name>
<name>
<surname>Bi</surname>
<given-names>X</given-names>
</name>
</person-group>. <article-title>Design and dsp implementation of a hyperchaotic map with infinite coexisting attractors and intermittent chaos based on a novel locally active memcapacitor</article-title>. <source>Chaos, Solitons and Fractals</source> (<year>2023</year>) <volume>173</volume>:<fpage>113708</fpage>. <pub-id pub-id-type="doi">10.1016/j.chaos.2023.113708</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Banerjee</surname>
<given-names>S</given-names>
</name>
</person-group>. <article-title>A simple photosensitive circuit based on a mutator for emulating memristor, memcapacitor, and meminductor: light illumination effects on dynamical behaviors</article-title>. <source>Int J Bifurcation Chaos</source> (<year>2024</year>) <volume>34</volume>:<fpage>2450069</fpage>. <pub-id pub-id-type="doi">10.1142/S021812742450069X</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>S</given-names>
</name>
<etal/>
</person-group> <article-title>A class of n-d Hamiltonian conservative chaotic systems with three-terminal memristor: modeling, dynamical analysis, and fpga implementation</article-title>. <source>Chaos</source> (<year>2025</year>) <volume>35</volume>:<fpage>013121</fpage>. <pub-id pub-id-type="doi">10.1063/5.0238893</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>D</given-names>
</name>
<name>
<surname>He</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>H</given-names>
</name>
</person-group>. <article-title>Resonant tunneling diode cellular neural network with memristor coupling and its application in police forensic digital image protection</article-title>. <source>Chin Phys B</source> (<year>2025</year>) <volume>34</volume>:<fpage>050502</fpage>. <pub-id pub-id-type="doi">10.1088/1674-1056/adb8bb</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>F</given-names>
</name>
<name>
<surname>He</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Q</given-names>
</name>
</person-group>. <article-title>Quantitative characterization system for macroecosystem attributes and states</article-title>. <source>IEEE Trans Computer-Aided Des Integrated Circuits Syst</source> (<year>2025</year>) <volume>36</volume>:<fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.13287/j.1001-9332.202501.031</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Banerjee</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>Hybrid image encryption scheme based on hyperchaotic map with spherical attractors</article-title>. <source>Chin Phys B</source> (<year>2025</year>) <volume>34</volume>:<fpage>030503</fpage>. <pub-id pub-id-type="doi">10.1088/1674-1056/ada7db</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>A novel multimodal joint information encryption scheme based on multi-level confusion and hyperchaotic map</article-title>. <source>Int J Mod Phys C</source> (<year>2025</year>). <pub-id pub-id-type="doi">10.1142/S012918312550038X</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Erkan</surname>
<given-names>U</given-names>
</name>
<name>
<surname>Tokta&#x15f;</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Novel hyperchaotic system: implementation to audio encryption</article-title>. <source>Chaos, Solitons and Fractals</source> (<year>2025</year>) <volume>193</volume>:<fpage>116088</fpage>. <pub-id pub-id-type="doi">10.1016/j.chaos.2025.116088</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Gracia</surname>
<given-names>YM</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>H</given-names>
</name>
</person-group>. <article-title>Dynamic analysis and implementation of fpga for a new 4d fractional-order memristive hopfield neural network</article-title>. <source>Fractal and Fractional</source> (<year>2025</year>) <volume>9</volume>:<fpage>115</fpage>. <pub-id pub-id-type="doi">10.3390/fractalfract9020115</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Mosaic tracking: lightweight batch video frame awareness multi-target encryption scheme based on a novel discrete tabu learning neuron and yolov5</article-title>. <source>IEEE Internet Things J</source> (<year>2024</year>) <volume>12</volume>:<fpage>4038</fpage>&#x2013;<lpage>49</lpage>. <pub-id pub-id-type="doi">10.1109/JIOT.2024.3482289</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Multi-face image compression encryption scheme combining extraction with stp-cs for face database</article-title>. <source>IEEE Internet Things J</source> (<year>2025</year>) <volume>12</volume>:<fpage>19522</fpage>&#x2013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1109/JIOT.2025.3541228</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Banerjee</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Combining semi-tensor product compressed sensing and session keys for low-cost encryption of batch information in wbans</article-title>. <source>IEEE Internet Things J</source> (<year>2024</year>) <volume>11</volume>:<fpage>33565</fpage>&#x2013;<lpage>76</lpage>. <pub-id pub-id-type="doi">10.1109/jiot.2024.3429349</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Iu</surname>
<given-names>HHC</given-names>
</name>
<name>
<surname>Erkan</surname>
<given-names>U</given-names>
</name>
<name>
<surname>Toktas</surname>
<given-names>A</given-names>
</name>
</person-group>. <article-title>Novel n-dimensional nondegenerate discrete hyperchaotic map with any desired lyapunov exponents</article-title>. <source>IEEE Internet Things J</source> (<year>2025</year>) <volume>12</volume>:<fpage>9082</fpage>&#x2013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1109/JIOT.2025.3541229</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>A novel memristor-coupled discrete neural network with multi-stability and multiple state transitions</article-title>. <source>Eur Phys J Spec Top</source> (<year>2025</year>). <pub-id pub-id-type="doi">10.1140/epjs/s11734-024-01440-8</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>Novel discrete initial-boosted tabu learning neuron: dynamical analysis, dsp implementation, and batch medical image encryption</article-title>. <source>Appl Intelligence</source> (<year>2025</year>) <volume>55</volume>:<fpage>61</fpage>. <pub-id pub-id-type="doi">10.1007/s10489-024-05918-9</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Banerjee</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Analysis of the functional behavior of fractional-order discrete neuron under electromagnetic radiation</article-title>. <source>Chaos, Solitons and Fractals</source> (<year>2023</year>) <volume>176</volume>:<fpage>114113</fpage>. <pub-id pub-id-type="doi">10.1016/j.chaos.2023.114113</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>A fhn-hr neuron network coupled with a novel locally active memristor and its dsp implementation</article-title>. <source>IEEE Trans Cybernetics</source> (<year>2024</year>) <volume>54</volume>:<fpage>7333</fpage>&#x2013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1109/TCYB.2024.3471644</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Mou</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Banerjee</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Multi-cube encryption scheme for multi-type images based on modified klotski game and hyperchaotic map</article-title>. <source>Nonlinear Dyn</source> (<year>2024</year>) <volume>112</volume>:<fpage>5727</fpage>&#x2013;<lpage>47</lpage>. <pub-id pub-id-type="doi">10.1007/s11071-024-09292-6</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Reich</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Prangemeier</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Cetin</surname>
<given-names>&#xd6;</given-names>
</name>
<name>
<surname>Koeppl</surname>
<given-names>H</given-names>
</name>
</person-group>. <article-title>Oss-net: memory efficient high resolution semantic segmentation of 3d medical data</article-title>. In: <source>Proceedings of the British machine vision conference (BMVC)</source> (<year>2021</year>). p. <fpage>429</fpage>.</citation>
</ref>
<ref id="B45">
<label>45.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khan</surname>
<given-names>MO</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Implicit neural representations for medical imaging segmentation. Medical image computing and computer assisted intervention &#x2013; MICCAI 2022 springer</article-title>. <source>Lecture Notes in Computer Sci</source> (<year>2022</year>) <volume>13431</volume>:<fpage>433</fpage>&#x2013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-16443-9_42</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Sapkota</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>DZ</given-names>
</name>
</person-group>. <article-title>SwIPE: efficient and robust medical image segmentation with implicit patch embeddings</article-title>. In: <source>Medical image computing and computer-assisted intervention &#x2013; MICCAI 2023</source>. <publisher-name>Springer Nature Switzerland</publisher-name> (<year>2023</year>). p. <fpage>315</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-43904-9_31</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mildenhall</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Srinivasan</surname>
<given-names>PP</given-names>
</name>
<name>
<surname>Tancik</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Barron</surname>
<given-names>JT</given-names>
</name>
<name>
<surname>Ramamoorthi</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>R</given-names>
</name>
</person-group>. <article-title>Nerf: representing scenes as neural radiance fields for view synthesis</article-title>. <source>Commun ACM</source> (<year>2022</year>) <volume>65</volume>:<fpage>99</fpage>&#x2013;<lpage>106</lpage>. <pub-id pub-id-type="doi">10.1145/3503250</pub-id>
</citation>
</ref>
<ref id="B48">
<label>48.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>S&#xf8;rensen</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Camara</surname>
<given-names>O</given-names>
</name>
<name>
<surname>Backer</surname>
<given-names>OD</given-names>
</name>
<name>
<surname>Kofoed</surname>
<given-names>KF</given-names>
</name>
<name>
<surname>Paulsen</surname>
<given-names>RR</given-names>
</name>
</person-group>. <article-title>NUDF: neural unsigned distance fields for high resolution 3d medical image segmentation</article-title>. In: <source>Proceedings of the 19th IEEE international symposium on biomedical imaging (ISBI)</source> (<year>2022</year>). p. <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/ISBI52829.2022.9761610</pub-id>
</citation>
</ref>
<ref id="B49">
<label>49.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Stolt-Ans&#xf3;</surname>
<given-names>N</given-names>
</name>
<name>
<surname>McGinnis</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Hammernik</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Rueckert</surname>
<given-names>D</given-names>
</name>
</person-group>. <article-title>Nisf: neural implicit segmentation functions</article-title>. In: <conf-name>Medical Image Computing and Computer-Assisted Intervention &#x2013; MICCAI 2023</conf-name>, <volume>14231</volume>. <publisher-loc>Vancouver, BC, Canada</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2023</year>). p. <fpage>734</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-43901-8_70</pub-id>
</citation>
</ref>
<ref id="B50">
<label>50.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Wickramasinghe</surname>
<given-names>U</given-names>
</name>
<name>
<surname>Ni</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Fua</surname>
<given-names>P</given-names>
</name>
</person-group>. <article-title>Implicitatlas: learning deformable shape templates in medical imaging</article-title>. In: <source>
<italic>Proceedings of the IEEE/CVF Conference on computer Vision and pattern recognition (CVPR)</italic> (new Orleans, LA, USA: IEEE)</source> (<year>2022</year>). p. <fpage>15861</fpage>&#x2013;<lpage>71</lpage>.</citation>
</ref>
<ref id="B51">
<label>51.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Molaei</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Aminimehr</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Tavakoli</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Kazerouni</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Azad</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Azad</surname>
<given-names>R</given-names>
</name>
<etal/>
</person-group> <article-title>Implicit neural representation in medical imaging: a comparative survey</article-title>. In: <source>
<italic>Proceedings of the IEEE/CVF international Conference on computer vision workshops (ICCVW)</italic> (paris, France: IEEE)</source> (<year>2023</year>). p. <fpage>2381</fpage>&#x2013;<lpage>91</lpage>. <pub-id pub-id-type="doi">10.1109/ICCVW60793.2023.00252</pub-id>
</citation>
</ref>
<ref id="B52">
<label>52.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Amiranashvili</surname>
<given-names>T</given-names>
</name>
<name>
<surname>L&#xfc;dke</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>HB</given-names>
</name>
<name>
<surname>Menze</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Zachow</surname>
<given-names>S</given-names>
</name>
</person-group>. <article-title>Learning shape reconstruction from sparse measurements with neural implicit functions</article-title>. In: <source>
<italic>Proceedings of the 5th international Conference on medical Imaging with deep learning (MIDL)</italic> (Z&#xfc;rich, Switzerland: PMLR)</source>, <volume>172</volume> (<year>2022</year>). p. <fpage>22</fpage>&#x2013;<lpage>34</lpage>.</citation>
</ref>
<ref id="B53">
<label>53.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chibane</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Alldieck</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Pons-Moll</surname>
<given-names>G</given-names>
</name>
</person-group>. <article-title>Implicit functions in feature space for 3d shape reconstruction and completion</article-title>. In: <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition (CVPR)</source>. <publisher-loc>Seattle, WA, USA</publisher-loc>: <publisher-name>IEEE</publisher-name> (<year>2020</year>). p. <fpage>6968</fpage>&#x2013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR42600.2020.00698</pub-id>
</citation>
</ref>
<ref id="B54">
<label>54.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bui</surname>
<given-names>NT</given-names>
</name>
<name>
<surname>Hoang</surname>
<given-names>DH</given-names>
</name>
<name>
<surname>Tran</surname>
<given-names>MT</given-names>
</name>
<name>
<surname>Doretto</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Adjeroh</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>B</given-names>
</name>
<etal/>
</person-group> <article-title>Sam3d: segment anything model in volumetric medical images</article-title>. In: <conf-name>Proceedings of the 2024 IEEE International Symposium on Biomedical Imaging (ISBI)</conf-name>. <publisher-name>Athens, Greece: IEEE</publisher-name> (<year>2024</year>). p. <fpage>1</fpage>&#x2013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B55">
<label>55.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S</given-names>
</name>
</person-group>. <article-title>I-medsam: implicit medical image segmentation with segment anything</article-title>. In: <source>
<italic>Proceedings of the European Conference on computer vision (ECCV)</italic> (Milan, Italy: Springer</source>, <volume>15068</volume> (<year>2024</year>). p. <fpage>90</fpage>&#x2013;<lpage>107</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-72684-2_6</pub-id>
</citation>
</ref>
<ref id="B56">
<label>56.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Landman</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Iglesias</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Styner</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Langerak</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Klein</surname>
<given-names>A</given-names>
</name>
</person-group>. <article-title>Miccai multi-atlas labeling beyond the cranial vault&#x2014;workshop and challenge</article-title>. <source>
<italic>Proc MICCAI Multi-Atlas Labeling Beyond Cranial Vault&#x2014;Workshop Challenge</italic> (Munich, Germany)</source> (<year>2015</year>) <volume>5</volume>:<fpage>12</fpage>.</citation>
</ref>
<ref id="B57">
<label>57.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Ge</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R</given-names>
</name>
<etal/>
</person-group> <article-title>Amos: a large-scale abdominal multi-organ benchmark for versatile medical image segmentation</article-title>. <source>Adv in Neural Inf Process Syst</source> (<year>2022</year>) <volume>35</volume>:<fpage>36722</fpage>&#x2013;<lpage>32</lpage>.</citation>
</ref>
<ref id="B58">
<label>58.</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Loshchilov</surname>
<given-names>I</given-names>
</name>
<name>
<surname>Hutter</surname>
<given-names>F</given-names>
</name>
</person-group>. <article-title>Decoupled weight decay regularization</article-title>. In: <conf-name>Proceedings of the 7th international Conference on learning representations (ICLR)</conf-name> (<publisher-loc>New Orleans, LA, USA</publisher-loc>) (<year>2019</year>).</citation>
</ref>
</ref-list>
</back>
</article>