<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Physiol.</journal-id>
<journal-title>Frontiers in Physiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Physiol.</abbrev-journal-title>
<issn pub-type="epub">1664-042X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1617647</article-id>
<article-id pub-id-type="doi">10.3389/fphys.2025.1617647</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>GA-TongueNet: tongue image segmentation network using innovative DiFP and MDi for stable generalization ability</article-title>
<alt-title alt-title-type="left-running-head">Dong et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphys.2025.1617647">10.3389/fphys.2025.1617647</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Dong</surname>
<given-names>Zhiyu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3039730/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhao</surname>
<given-names>Le</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3039733/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Fan</surname>
<given-names>Yajun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ma</surname>
<given-names>Haihua</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Shao</surname>
<given-names>Changle</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Yiran</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Li</surname>
<given-names>Peng</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Information Science and Engineering</institution>, <institution>Henan University of Technology</institution>, <addr-line>Zhengzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Institute for Complexity Science</institution>, <institution>Henan University of Technology</institution>, <addr-line>Zhengzhou</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/532776/overview">Rajesh Kumar Tripathy</ext-link>, Birla Institute of Technology and Science, India</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/711875/overview">Nguyen Quoc Khanh Le</ext-link>, Taipei Medical University, Taiwan</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2097715/overview">Mengjian Zhang</ext-link>, South China University of Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Le Zhao, <email>lezhao@haut.edu.cn</email>; Peng Li, <email>lipeng@haut.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>24</day>
<month>06</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1617647</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Dong, Zhao, Fan, Ma, Shao, Zhang and Li.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Dong, Zhao, Fan, Ma, Shao, Zhang and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Tongue is directly or indirectly connected to many internal organs in Traditional Chinese Medicine (TCM). In computer-aided diagnosis, tongue image segmentation is the first step in tongue diagnosis, and the precision of this segmentation is decisive in determining the accuracy of the tongue diagnosis results. Due to challenges such as insufficient available sample size and complex background, the generalization and robustness of current tongue segmentation algorithms are usually poor, which seriously hinders the practicality of tongue diagnosis. In this article, a GA-TongueNet, namely Tongue Segmentation Network for Stable Generalization Ability, based on self-attention architecture is proposed, which is a tongue segmentation network that can simultaneously have strong generalization ability and accuracy under small samples and diverse background conditions. Firstly, GA-TongueNet is built upon the transformer architecture, embedding the dilated feature pyramid (DiFP) module and the multi-dilated convolution (MDi) module proposed in this article. Secondly, the DiFP module is integrated to comprehend both the overall tongue image structure and intricate local details, while the MDi module is specifically designed to preserve a high feature resolution. Therefore, the network adeptly captures long-range dependencies, extracts high-level semantic content, and retains low-level detail information from tongue images. Moreover, it maintains decent precision and stable generalization capabilities, even when dealing with limited sample sizes. Experimental results show that the accuracy and generalization ability of GA-TongueNet in complex environments are significantly better than various existing semantic segmentation algorithms based on Convolutional Neural Networks (CNN) and Transformer architectures.</p>
</abstract>
<kwd-group>
<kwd>tongue segmentation</kwd>
<kwd>self-attention</kwd>
<kwd>transformer</kwd>
<kwd>dilated convolution</kwd>
<kwd>feature pyramid networks</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Physiology and Medicine</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>As a widely accepted complementary and alternative medical approach, TCM has garnered increasing attention from the medical research community (<xref ref-type="bibr" rid="B33">Xu et al., 2020</xref>). Within the four diagnostic methods of TCM, tongue diagnosis constitutes a pivotal component of the observation procedure, serving as a cornerstone for clinical evaluation. By observing various characteristics of the tongue, such as its shape, color, and coating, practitioners can assess health conditions, the nature of diseases, and the functional state of internal organs (<xref ref-type="bibr" rid="B31">Wang et al., 2020</xref>). Rooted in TCM theory, the tongue serves as an external reflection of qi (vital energy) and blood circulation within the visceral systems, often referred to as the &#x201c;visceral mirror&#x201d;. This diagnostic approach provides an effective and non-invasive method for health assessment (<xref ref-type="bibr" rid="B40">Zhang et al., 2025</xref>). However, traditional tongue diagnosis relies heavily on empirical knowledge and subjective judgment, which may limit its reliability (<xref ref-type="bibr" rid="B11">Hu et al., 2019</xref>). With the rapid advancement of artificial intelligence in the medical field (<xref ref-type="bibr" rid="B15">Le et al., 2019</xref>), computer-aided tongue diagnosis has emerged as a promising avenue for addressing these limitations (<xref ref-type="bibr" rid="B7">Gao et al., 2022</xref>).</p>
<p>Computer-aided tongue diagnosis models typically rely on training and analysis of images captured by specialized tongue image acquisition devices (<xref ref-type="bibr" rid="B3">Cai et al., 2024</xref>). However, these images often include irrelevant facial or device-related information. Additionally, the inherent limitations of the acquisition devices lead to poor adaptability in diverse scenarios (<xref ref-type="bibr" rid="B42">Zhou et al., 2022</xref>). These challenges result in deviations in feature extraction and undermine the diagnostic reliability of such models (<xref ref-type="bibr" rid="B25">Qiu et al., 2023</xref>). Therefore, the development of a robust tongue image segmentation algorithm first enables precise extraction of critical pathological parameters including tongue substance and tongue coating by effectively separating the tongue body from extraneous background noise (<xref ref-type="bibr" rid="B4">Cao et al., 2023</xref>), and further serves as a crucial foundation for enhancing the accuracy and robustness of diagnostic models (<xref ref-type="bibr" rid="B38">Zhang et al., 2019</xref>).</p>
<p>The recent advancements in deep learning-related technologies have provided multiple research approaches for the tongue image segmentation task (<xref ref-type="bibr" rid="B29">Tng et al., 2022</xref>). Current mainstream methods are primarily based on CNN (<xref ref-type="bibr" rid="B41">Zhao et al., 2022</xref>), which leverage their powerful feature extraction capabilities to achieve notable success in semantic segmentation tasks and advance the field (<xref ref-type="bibr" rid="B35">Yu et al., 2021</xref>). However, CNN encounter inherent limitations when applied to complex natural environments (<xref ref-type="bibr" rid="B23">Monica et al., 2024</xref>). The restricted receptive field of convolution operations hinders their capacity to effectively capture contextual information (<xref ref-type="bibr" rid="B17">Liang et al., 2023</xref>). This limitation complicates the understanding of overall semantics and spatial relationships in tongue images, especially under diverse and challenging conditions (<xref ref-type="bibr" rid="B6">Feng et al., 2021</xref>). These limitations lead to poor generalization when dealing with tongue image data captured in varying acquisition environments, lighting conditions, and shooting angles. Additionally, CNN often struggle to accurately delineate the subtle edges of the tongue, resulting in segmentation precision that falls short of practical requirements (<xref ref-type="bibr" rid="B16">Li et al., 2021</xref>). The transformer architectures can effectively establish an integration mechanism for both local and global contextual information through self-attention mechanisms (<xref ref-type="bibr" rid="B39">Zhang et al., 2023</xref>), addressing the limitations of CNN and enhancing the model&#x2019;s feature representation capabilities. However, these models impose substantial computational demands and require large-scale labeled datasets to prevent overfitting, as insufficient data often results in unstable convergence and compromised generalization performance.</p>
<p>To address these challenges, this article proposes GA-TongueNet, a novel model designed to enhance the generalization ability and stability of tongue image segmentation. Even when trained solely on datasets captured under standard acquisition scenarios, GA-TongueNet demonstrates high-precision boundary positioning capabilities and robust segmentation performance across diverse lighting conditions in natural environments. The key contributions of this article are summarized as follows:<list list-type="simple">
<list-item>
<p>&#x2022; GA-TongueNet is proposed for tongue image semantic segmentation, effectively capturing long-range dependencies, high-level semantic information, and low-level detail information. The network achieves high precision and robust generalization even under complex backgrounds and small sample sizes.</p>
</list-item>
<list-item>
<p>&#x2022; A novel DiFP module is proposed to better capture the overall structure and local details of tongue images, while the MDi module is designed to handle tongue images of varying sizes and maintain high feature resolution.</p>
</list-item>
<list-item>
<p>&#x2022; Notably, GA-TongueNet achieves superior cross-domain generalization compared to Masked Autoencoders (MAE)-based methods without requiring pre-training, highlighting its inherent ability to learn discriminative features from limited data.</p>
</list-item>
</list>
</p>
<p>The remainder of this article is organized as follows: <xref ref-type="sec" rid="s2">Section 2</xref> reviews related work on tongue image segmentation. <xref ref-type="sec" rid="s3">Section 3</xref> details the proposed method, experimental materials and experimental results. <xref ref-type="sec" rid="s5">Section 5</xref> discusses the experimental results. Finally, <xref ref-type="sec" rid="s6">Section 6</xref> concludes the article.</p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<sec id="s2-1">
<title>2.1 Clustering-based methods</title>
<p>Clustering algorithms offer a promising approach for tongue image segmentation due to their ability to operate without heavy reliance on labeled data (<xref ref-type="bibr" rid="B37">Yuan et al., 2023</xref>). By automatically extracting and summarizing image features, they provide a flexible and versatile solution for segmenting biologically relevant structures in tongue images. For instance, to facilitate the automatic diagnosis of tongue images, <xref ref-type="bibr" rid="B8">Guo et al. (2016)</xref> proposed an automatic region segmentation algorithm that combines K-Means clustering with an adaptive activity contour network. <xref ref-type="bibr" rid="B19">Liu et al. (2018)</xref> improved the SLIC gamut distance formula, making the superpixels generated by SLIC more suitable for tongue image segmentation and reducing the segmentation time of the Grab Cut method. SGSCN (<xref ref-type="bibr" rid="B1">Ahn et al., 2021</xref>) iteratively learns the feature representation and cluster assignment for each pixel within a single image, while simultaneously ensuring that all pixels within a cluster remain spatially close to its center.</p>
<p>Despite their computational efficiency, clustering algorithms face inherent limitations in processing complex tongue images. The high variability in biological characteristics, such as texture, color, and morphology, can exceed the adaptability of these algorithms. Moreover, they are particularly vulnerable to noise interference, which may lead to substantial deviations in segmentation accuracy. These constraints diminish the reliability and robustness of clustering-based methods, making them less effective in addressing the nuanced demands of tongue image segmentation for biomedical applications.</p>
</sec>
<sec id="s2-2">
<title>2.2 CNN-based methods</title>
<p>CNN demonstrate strong feature extraction capabilities in processing tongue images, making them valuable for analyzing biological characteristics such as texture and color while preserving intricate details (<xref ref-type="bibr" rid="B20">Liu et al., 2022</xref>). For example, OET-NET (<xref ref-type="bibr" rid="B13">Huang et al., 2022</xref>) incorporates a residual soft connection module and a prominent image fusion module, coupled with a Focal Loss-based optimization strategy, to achieve effective tongue image segmentation in controlled environments. To address challenges associated with small sample sizes, QA-TSN (<xref ref-type="bibr" rid="B14">Jia et al., 2025</xref>) introduces a global rendering block to enhance global feature representation and employs modified partial convolution to accelerate real-time segmentation. Similarly, LAIU-Net (<xref ref-type="bibr" rid="B22">Marhamati et al., 2023</xref>) applies an optimized data augmentation strategy to segment biologically complex structures, such as sunken human tongues in photographic images. HPA-UNet (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>) improves segmentation accuracy through enhanced data augmentation techniques and an updated U-Net architecture.</p>
<p>However, CNN-based methods encounter inherent limitations in addressing the complexities of tongue image segmentation, particularly when dealing with biological variability and challenging environmental conditions. These challenges include difficulty in segmenting small, biologically relevant structures, limited ability to capture global contextual features, and reduced generalization capabilities across diverse scenarios. Such limitations highlight the need for more robust and adaptable approaches to advance the segmentation of tongue images for biomedical applications.</p>
</sec>
<sec id="s2-3">
<title>2.3 Transformer-based methods</title>
<p>Transformers, with their self-attention mechanisms, offer a powerful framework for modeling relationships among different regions within an image. This characteristic is particularly beneficial for tongue image segmentation, where the accurate delineation of the tongue region is essential for analyzing biological features. Additionally, Transformers&#x2019; capacity for feature fusion enables the integration of multi-level information, enhancing segmentation accuracy. For instance, PriTongueNet (<xref ref-type="bibr" rid="B12">Huang et al., 2025</xref>) incorporates attention-guided skip connections and a self-distillation mechanism to address over-segmentation by supervising feature map differences during training. Similarly, Tongue-LiteSAM (<xref ref-type="bibr" rid="B27">Tan et al., 2025</xref>), a zero-shot model, achieves segmentation by integrating lightweight ViT-Tiny models based on the Segment Anything Model, providing a flexible approach for tongue image analysis. To minimize noise interference, Polyp-PVT (<xref ref-type="bibr" rid="B30">Wan et al., 2024</xref>) leverages the Swin-Transformer&#x2019;s advanced feature extraction capabilities for analyzing sublingual veins, demonstrating its potential in capturing subtle biological details. In broader medical image segmentation, Slim UNETR (<xref ref-type="bibr" rid="B24">Pang et al., 2024</xref>) employs a decomposed self-attention mechanism to efficiently aggregate representations, achieving robust performance on resource-constrained devices.</p>
<p>While having advantages in capturing global features, transformers face limitations in modeling localized biological details. Their integration with other networks often encounters challenges in effectively balancing high-level semantic information with low-level structural details. Furthermore, Transformer-based architectures typically require large-scale training datasets to prevent overfitting. When applied to small-sample datasets, such as those often encountered in tongue image segmentation, the models may suffer from reduced generalization capability. This limitation is particularly pronounced in scenarios involving discontinuous tongue edges or complex backgrounds, where achieving fine-grained and biologically accurate segmentation remains a significant challenge.</p>
</sec>
</sec>
<sec sec-type="materials|methods" id="s3">
<title>3 Materials and methods</title>
<sec id="s3-1">
<title>3.1 Tongue segmentation datasets</title>
<p>The datasets utilized in this article consist of two subsets: Dataset A and Dataset B. Dataset A is used for both training and testing the model, while Dataset B is solely employed to evaluate the model&#x2019;s generalization capability. Sample images from both datasets are shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Sample images from the datasets. <bold>(A&#x2013;C)</bold> are from Dataset A, <bold>(D&#x2013;F)</bold> are from Dataset B.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g001.tif">
<alt-text content-type="machine-generated">Two datasets, A and B, show multiple photographs of tongues. Dataset A includes standardized images of tongues from different angles, some with a coating and variations in texture. Dataset B features more casual images displaying tongues with various colors, textures, and lighting conditions.</alt-text>
</graphic>
</fig>
<sec id="s3-1-1">
<title>3.1.1 Dataset A</title>
<p>Dataset A originates from the publicly available BioHit (<xref ref-type="bibr" rid="B2">BioHit, 2014</xref>) dataset, comprising 300 tongue images with a resolution of <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mn>768</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>576</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> pixels. All images were collected using a standardized tongue imaging device, ensuring consistency in the positioning of the tongue across images. Corresponding ground truth annotations were meticulously prepared by experienced professionals, guaranteeing high-quality labeled data for model training.</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Dataset B</title>
<p>To further evaluate the generalization performance of the proposed model, we produced Dataset B by collecting 100 tongue images from public service tongue diagnosis posts on different online platforms. Tongue images were collected following stringent criteria to ensure diversity and realism in Dataset B. The collection guidelines excluded the use of filters, beauty enhancements, and identifiable features such as full facial images. Additionally, the dataset incorporates a variety of tongue-to-image size proportions, diverse lighting conditions, and multiple acquisition environments. These measures were designed to closely approximate real-world scenarios and enhance the dataset&#x2019;s representativeness.</p>
<p>The labeling process utilized Labelme 5.5.0 to perform detailed segmentation annotations of the tongue body. According to TCM theory, different regions of the tongue correspond to various organs of the body. Therefore, the labeling approach adhered to a comprehensive standard, aiming to annotate all discernible parts of the tongue within the images. This included challenging areas such as the tongue root, which is often under-illuminated within the oral cavity. Efforts were made to ensure precise and thorough annotations, even in less visible regions.</p>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 The proposed method</title>
<p>In this section, the GA-TongueNet will be elaborated in detail. GA-TongueNet is inspired by the architecture of SegFormer (<xref ref-type="bibr" rid="B32">Xie et al., 2021</xref>) and employs a transformer-based encoder-decoder framework. Specifically, to ensure the trainability, convergence, and generalization ability of the model under small-sample conditions, we propose DiFP and MDi to enhance the model&#x2019;s capability in perceiving and representing local detail features, global features, and contextual information.</p>
<sec id="s3-2-1">
<title>3.2.1 The architecture of GA-TongueNet</title>
<p>The model proposed in this article comprises three key modules, as illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>: the backbone for feature extraction, the neck for feature fusion, and the head for prediction. The backbone is based on the architecture of SegFormer and fully leverages the MixVision Transformer (MIT) module&#x2019;s capability to extract multi-scale features.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>GA-TongueNet architecture.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g002.tif">
<alt-text content-type="machine-generated">Diagram showing a neural network architecture for tongue image processing. The system includes a Backbone with a MixVision Transformer, a Neck with DiFP, and a Head consisting of a Di MLP Layer and MLP layers, leading to tongue image segmentation. Input images of tongues on the left transform to segmented images on the right, highlighting specific areas. Mathematical equations and diagrams detail convolutional layers and operations within the process.</alt-text>
</graphic>
</fig>
<p>For the neck module, we introduce targeted structural enhancements, resulting in the development of the DiFP module. This module efficiently fuses multi-scale feature information and, more importantly, deeply captures contextual information without compromising resolution. Such capabilities significantly contribute to precise segmentation of tongue details, thereby improving both the granularity and generalization ability of the segmentation results.</p>
<p>In the head module, we innovatively propose the MDi and incorporate it into the MLP layer. This design allows the model to develop a deeper understanding of the global structure within images, enabling it to effectively handle tongue images of varying sizes. Consequently, this enhances the model&#x2019;s adaptability and generalization performance, ensuring accurate and robust segmentation across diverse scenarios.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 The construction of the DiFP module</title>
<p>The Feature Pyramid Networks (FPN) (<xref ref-type="bibr" rid="B18">Lin et al., 2017</xref>) have demonstrated outstanding capabilities in processing multi-scale features. However, its performance encounters certain limitations when applied to precise pixel-level prediction tasks. To construct a high-performance multi-level feature map structure capable of efficiently handling multi-scale objects while maintaining high feature map resolution, we meticulously improved the original FPN module to develop the DiFP, which serves as the neck component of our model.</p>
<p>Building upon the original FPN, the DiFP integrates the key technology of dilated convolution. Specifically, <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> dilated convolutions with varying dilation rates <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> replace the original standard <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution operations, thereby enhancing the smoothness of the feature maps. <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates the input and output feature maps of the module. The core innovation of dilated convolution lies in its ability to expand the receptive field effectively without significantly increasing the number of parameters, achieved by strategically inserting gaps between the convolution kernel elements. This is further explained in <xref ref-type="disp-formula" rid="e1">Formula 1</xref> for the receptive field.<disp-formula id="e1">
<mml:math id="m5">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mspace width="0.3333em"/>
<mml:mi>F</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the size of the convolution kernel, and <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the different dilation rates.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The input and output feature maps of DiFP.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g003.tif">
<alt-text content-type="machine-generated">Diagram showing a flowchart of tongue images processed through a series of convolutional operations. The images on the left are input, progressing through blocks labeled DiConv1, DiConv2, DiConv4, and DiConv6, resulting in the output images on the right. Color variations highlight different processing stages.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="disp-formula" rid="e2">Formula 2</xref> ensures that the spatial resolution of the output feature map remains consistent with the input after the application of dilated convolution, which is critical for preserving resolution in pixel-level prediction tasks. In the proposed DiFP, the dilation rates <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are set to 1, 2, 4, and 6 to effectively capture features at varying receptive fields. To counterbalance the increased spatial requirements introduced by dilated convolutions, the padding <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is configured to match the dilation rate <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Furthermore, with the kernel size <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> fixed at 3 and stride <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> set to 1, the structural integrity and resolution consistency of the feature maps are maintained throughout the process.<disp-formula id="e2">
<mml:math id="m13">
<mml:mrow>
<mml:mfenced open="{" close="">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="&#x230a;" close="&#x230b;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>p</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="&#x230a;" close="&#x230b;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>p</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where, <inline-formula id="inf12">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf13">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the height and width of the output feature map, respectively, while <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the height and width of the input feature map. The parameter <inline-formula id="inf16">
<mml:math id="m18">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> corresponds to the padding applied, <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> refers to the stride, and <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:mo>&#x230a;</mml:mo>
<mml:mo>&#x230b;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> indicates the downward rounding operation.</p>
<p>The DiFP method effectively captures the target&#x2019;s multi-scale features, enabling the processing of objects at varying scales. Additionally, it preserves the resolution of the feature map, thereby enhancing prediction accuracy in fine-grained tasks while ensuring robust generalization performance across diverse data distributions.</p>
</sec>
<sec id="s3-2-3">
<title>3.2.3 The construction of the MDi module</title>
<p>To enable the model to accurately capture contextual information at multiple scales and extract finer details, we developed the MDi module and integrated it into the Di MLP layer. By combining dilated convolutions with both large and small dilation rates while retaining the original feature map, this design effectively enhances the model&#x2019;s ability to capture both local and global features. With a simple structure that relies on basic connections and <inline-formula id="inf19">
<mml:math id="m21">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolutions for feature fusion, the MDi module improves computational efficiency, enhances sensitivity to complex scenes, and boosts performance in boundary detection tasks.</p>
<p>As illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>, the MDi module employs a multi-scale dilated convolution mechanism to generate multi-scale feature maps. Specifically, each input feature map is processed through <inline-formula id="inf20">
<mml:math id="m22">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> dilated convolution layers, resulting in <inline-formula id="inf21">
<mml:math id="m23">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> feature maps, where the additional map corresponds to the original input feature. The colored grid in <xref ref-type="fig" rid="F4">Figure 4</xref> represents the multi-scale receptive field, demonstrating the module&#x2019;s ability to balance local and global feature extraction effectively. Dilated convolutions with smaller dilation rates excel at capturing fine-grained boundary details, ensuring that subtle features are preserved. In contrast, larger dilation rates allow the module to extract broader contextual information, contributing to a more comprehensive understanding of the overall scene.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Multi-scale dilated convolution in MDi module. RF represents receptive fields.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g004.tif">
<alt-text content-type="machine-generated">Diagram illustrating different dilation rates and receptive fields on a spatial grid. Colored squares represent dilation rates: red for \( d = 1 \) (RF \(3 \times 3\)), blue for \( d = 2 \) (RF \(5 \times 5\)), green for \( d = 4 \) (RF \(9 \times 9\)), and purple for \( d = 6 \) (RF \(13 \times 13\)). The central area is highlighted with a circle.</alt-text>
</graphic>
</fig>
<p>After multi-scale feature fusion, the MDi module produces a combined feature map, mathematically expressed in <xref ref-type="disp-formula" rid="e3">Equation 3</xref>. This fused feature map integrates local details with global structural features of the tongue image. The fusion process, visually represented as a colored grid overlay, highlights how the expanded receptive field enables accurate and comprehensive segmentation. By achieving a balanced representation of fine and coarse features, the MDi module significantly improves the model&#x2019;s segmentation performance.<disp-formula id="e3">
<mml:math id="m24">
<mml:mrow>
<mml:mi>Z</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf22">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the result of the <italic>i</italic>th input feature map after passing through a <inline-formula id="inf23">
<mml:math id="m26">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution and upsampling. <inline-formula id="inf24">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the result of the <italic>i</italic>th input feature map after passing through the <italic>j</italic>th dilated convolution. <inline-formula id="inf25">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the weight matrix for the <inline-formula id="inf26">
<mml:math id="m29">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution used for feature fusion, <inline-formula id="inf27">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the bias term for the <inline-formula id="inf28">
<mml:math id="m31">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> convolution, and <inline-formula id="inf29">
<mml:math id="m32">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the activation function.</p>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Implementation details</title>
<p>The models utilized in this study were developed and implemented within the framework of mmsegmentation 1.2.2, with the exception of RTC_TongueNet (<xref ref-type="bibr" rid="B28">Tang et al., 2024</xref>) and TongueSAM (<xref ref-type="bibr" rid="B4">Cao et al., 2023</xref>), which were evaluated using their officially recommended configuration environments to ensure optimal performance. All other models were implemented using Python 3.8.2 and PyTorch 2.1.0, with computations powered by CUDA 11.8 on an ASUS TUF Gaming FX507VV platform (CPU: Intel Core i7-13700H; GPU: NVIDIA GeForce RTX 4060, 8 GB). The Dataset A, comprising 300 tongue images, was randomly divided into a training set (270 images) and a test set (30 images). During training, a variety of data augmentation techniques were applied to enhance the model&#x2019;s robustness and adaptability to different input conditions. These techniques included scaling (resizing with a factor range of 0.5&#x2013;2.0), cropping (retaining a random 75% area of the image), flipping with a probability of 0.5, and adjusting brightness (ranging from &#x2212;32 to &#x2b;32), contrast (ranging from 0.5 to 1.5), and saturation (ranging from 0.5 to 1.5). These augmentation strategies ensured that the model was exposed to a wide range of variations during training, enhancing its robustness and generalization performance across diverse datasets and input scenarios.</p>
</sec>
</sec>
<sec sec-type="results" id="s4">
<title>4 Results</title>
<sec id="s4-1">
<title>4.1 Evaluation metrics</title>
<p>To assess the performance of the proposed model, four commonly used tongue segmentation evaluation metrics are employed: Dice, IoU, Precision, and Recall. Dice, as shown in <xref ref-type="disp-formula" rid="e4">Equation 4</xref>, denotes the metric of overlap between two sets, reflecting the accuracy of tongue target extraction. IoU, as shown in <xref ref-type="disp-formula" rid="e5">Equation 5</xref>, defined as the ratio of intersection to concatenation, intuitively reflects the accuracy of the segmentation results. Precision, as shown in <xref ref-type="disp-formula" rid="e6">Equation 6</xref>, refers to the ratio of the number of correctly predicted positive samples among all the predicted positive samples, which reflects the reliability of the prediction results. Recall, as shown in <xref ref-type="disp-formula" rid="e7">Equation 7</xref>, refers to the ratio of the number of correctly predicted positive samples to the total number of true positive samples, which evaluates the completeness of tongue segmentation. Higher values of the above four indicators mean better segmentation performance of the model. The formulas are as follows:<disp-formula id="e4">
<mml:math id="m33">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m34">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2229;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x222a;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf30">
<mml:math id="m35">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf31">
<mml:math id="m36">
<mml:mrow>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denote the sets of predicted and ground truth pixels, respectively.<disp-formula id="e6">
<mml:math id="m37">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m38">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf32">
<mml:math id="m39">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf33">
<mml:math id="m40">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf34">
<mml:math id="m41">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denote true positives, false positives and false negatives, respectively.</p>
</sec>
<sec id="s4-2">
<title>4.2 Methods comparison</title>
<p>In this study, the performance of the proposed GA-TongueNet was comprehensively compared with eight other models. These include six well-established semantic segmentation models: DeepLabV3plus (<xref ref-type="bibr" rid="B5">Chen et al., 2018</xref>), U-Net (<xref ref-type="bibr" rid="B26">Ronneberger et al., 2015</xref>), Swin Transformer (<xref ref-type="bibr" rid="B21">Liu et al., 2021</xref>), SegFormer (<xref ref-type="bibr" rid="B32">Xie et al., 2021</xref>), SegNeXt (<xref ref-type="bibr" rid="B9">Guo et al., 2022</xref>), and PoolFormer (<xref ref-type="bibr" rid="B36">Yu et al., 2022</xref>), as well as two recently developed models specifically designed for tongue image segmentation: RTC_TongueNet (<xref ref-type="bibr" rid="B28">Tang et al., 2024</xref>) and TongueSAM (<xref ref-type="bibr" rid="B4">Cao et al., 2023</xref>). The sizes and inference speeds of these models are summarized in <xref ref-type="table" rid="T1">Table 1</xref>. To evaluate their performance in a standard acquisition environment, all models were trained and tested on Dataset A. Additionally, to assess their generalization capability, further tests were conducted on Dataset B, which comprises more challenging and diverse scenarios. The evaluation metrics include Dice, IoU, Precision, and Recall. Quantitative results for both Dataset A and Dataset B are presented in <xref ref-type="table" rid="T2">Table 2</xref> and are visually summarized in the radar chart shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. Furthermore, representative segmentation outcomes from the two datasets are illustrated in <xref ref-type="fig" rid="F6">Figures 6</xref>, <xref ref-type="fig" rid="F7">7</xref>, providing qualitative comparisons that highlight the strengths and weaknesses of the various models.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Model size and inference speed of different networks.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Network</th>
<th align="left">Parameters (M)</th>
<th align="left">FPS</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">DeepLabV3plus (<xref ref-type="bibr" rid="B5">Chen et al., 2018</xref>)</td>
<td align="left">41.22</td>
<td align="left">13.35</td>
</tr>
<tr>
<td align="left">U-Net (<xref ref-type="bibr" rid="B26">Ronneberger et al., 2015</xref>)</td>
<td align="left">28.99</td>
<td align="left">3.84</td>
</tr>
<tr>
<td align="left">Swin Transformer (<xref ref-type="bibr" rid="B21">Liu et al., 2021</xref>)</td>
<td align="left">41.64</td>
<td align="left">11.50</td>
</tr>
<tr>
<td align="left">SegFormer (<xref ref-type="bibr" rid="B32">Xie et al., 2021</xref>)</td>
<td align="left">0.89</td>
<td align="left">60.33</td>
</tr>
<tr>
<td align="left">SegNeXt (<xref ref-type="bibr" rid="B9">Guo et al., 2022</xref>)</td>
<td align="left">4.26</td>
<td align="left">43.42</td>
</tr>
<tr>
<td align="left">PoolFormer (<xref ref-type="bibr" rid="B36">Yu et al., 2022</xref>)</td>
<td align="left">15.634</td>
<td align="left">58.79</td>
</tr>
<tr>
<td align="left">RTC_TongueNet (<xref ref-type="bibr" rid="B28">Tang et al., 2024</xref>)</td>
<td align="left">55.42</td>
<td align="left">5.99</td>
</tr>
<tr>
<td align="left">TongueSAM (<xref ref-type="bibr" rid="B4">Cao et al., 2023</xref>)</td>
<td align="left">102.67</td>
<td align="left">3.36</td>
</tr>
<tr>
<td align="left">Ours</td>
<td align="left">14.01</td>
<td align="left">11.77</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Evaluation metrics data for different networks.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Dataset</th>
<th align="left">Network</th>
<th align="left">Dice</th>
<th align="left">IoU</th>
<th align="left">Precision</th>
<th align="left">Recall</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="9" align="left">Dataset A</td>
<td align="left">DeepLabV3plus (<xref ref-type="bibr" rid="B5">Chen et al., 2018</xref>)</td>
<td align="left">0.9851</td>
<td align="left">0.9706</td>
<td align="left">0.9879</td>
<td align="left">0.9823</td>
</tr>
<tr>
<td align="left">U-Net (<xref ref-type="bibr" rid="B26">Ronneberger et al., 2015</xref>)</td>
<td align="left">0.9765</td>
<td align="left">0.9541</td>
<td align="left">0.9827</td>
<td align="left">0.9704</td>
</tr>
<tr>
<td align="left">Swin Transformer (<xref ref-type="bibr" rid="B21">Liu et al., 2021</xref>)</td>
<td align="left">0.9863</td>
<td align="left">0.9729</td>
<td align="left">0.9885</td>
<td align="left">0.9841</td>
</tr>
<tr>
<td align="left">SegFormer (<xref ref-type="bibr" rid="B32">Xie et al., 2021</xref>)</td>
<td align="left">0.9734</td>
<td align="left">0.9481</td>
<td align="left">
<bold>0.9909</bold>
</td>
<td align="left">0.9565</td>
</tr>
<tr>
<td align="left">SegNeXt (<xref ref-type="bibr" rid="B9">Guo et al., 2022</xref>)</td>
<td align="left">0.9890</td>
<td align="left">0.9781</td>
<td align="left">0.9873</td>
<td align="left">0.9906</td>
</tr>
<tr>
<td align="left">PoolFormer (<xref ref-type="bibr" rid="B36">Yu et al., 2022</xref>)</td>
<td align="left">0.9845</td>
<td align="left">0.9694</td>
<td align="left">0.9865</td>
<td align="left">0.9824</td>
</tr>
<tr>
<td align="left">RTC_TongueNet (<xref ref-type="bibr" rid="B28">Tang et al., 2024</xref>)</td>
<td align="left">0.9640</td>
<td align="left">0.9310</td>
<td align="left">0.9691</td>
<td align="left">0.9850</td>
</tr>
<tr>
<td align="left">TongueSAM (<xref ref-type="bibr" rid="B4">Cao et al., 2023</xref>)</td>
<td align="left">0.9476</td>
<td align="left">0.9005</td>
<td align="left">0.9141</td>
<td align="left">0.9883</td>
</tr>
<tr>
<td align="left">Ours</td>
<td align="left">
<bold>0.9906</bold>
</td>
<td align="left">
<bold>0.9814</bold>
</td>
<td align="left">0.9894</td>
<td align="left">
<bold>0.9918</bold>
</td>
</tr>
<tr>
<td rowspan="9" align="left">Dataset B</td>
<td align="left">DeepLabV3plus (<xref ref-type="bibr" rid="B5">Chen et al., 2018</xref>)</td>
<td align="left">0.8953</td>
<td align="left">0.8334</td>
<td align="left">0.8585</td>
<td align="left">0.9681</td>
</tr>
<tr>
<td align="left">U-Net (<xref ref-type="bibr" rid="B26">Ronneberger et al., 2015</xref>)</td>
<td align="left">0.8171</td>
<td align="left">0.7092</td>
<td align="left">0.8560</td>
<td align="left">0.8219</td>
</tr>
<tr>
<td align="left">Swin Transformer (<xref ref-type="bibr" rid="B21">Liu et al., 2021</xref>)</td>
<td align="left">0.8587</td>
<td align="left">0.7874</td>
<td align="left">0.8190</td>
<td align="left">0.9567</td>
</tr>
<tr>
<td align="left">SegFormer (<xref ref-type="bibr" rid="B32">Xie et al., 2021</xref>)</td>
<td align="left">0.9339</td>
<td align="left">0.8784</td>
<td align="left">0.9289</td>
<td align="left">0.9423</td>
</tr>
<tr>
<td align="left">SegNeXt (<xref ref-type="bibr" rid="B7">Guo et al., 2022</xref>)</td>
<td align="left">0.8757</td>
<td align="left">0.7895</td>
<td align="left">0.8103</td>
<td align="left">0.9697</td>
</tr>
<tr>
<td align="left">PoolFormer (<xref ref-type="bibr" rid="B36">Yu et al., 2022</xref>)</td>
<td align="left">0.9358</td>
<td align="left">0.8848</td>
<td align="left">0.9349</td>
<td align="left">0.9440</td>
</tr>
<tr>
<td align="left">RTC_TongueNet (<xref ref-type="bibr" rid="B28">Tang et al., 2024</xref>)</td>
<td align="left">0.4975</td>
<td align="left">0.6463</td>
<td align="left">0.5759</td>
<td align="left">0.8184</td>
</tr>
<tr>
<td align="left">TongueSAM (<xref ref-type="bibr" rid="B4">Cao et al., 2023</xref>)</td>
<td align="left">0.9356</td>
<td align="left">0.8790</td>
<td align="left">0.8893</td>
<td align="left">
<bold>0.9870</bold>
</td>
</tr>
<tr>
<td align="left">Ours</td>
<td align="left">
<bold>0.9553</bold>
</td>
<td align="left">
<bold>0.9163</bold>
</td>
<td align="left">
<bold>0.9501</bold>
</td>
<td align="left">0.9632</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values represent the optimal values.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Radar chart of evaluation metrics for different networks on datasets. <bold>(A)</bold> represents Dataset A and <bold>(B)</bold> represents Dataset B.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g005.tif">
<alt-text content-type="machine-generated">Two radar charts compare different models, labeled as A and B. Both charts measure Dice, IoU, Recall, and Precision with values ranging from 0 to 1. Models include DeepLabV3plus, U-Net, Swin Transformer, SegFormer, and others, each represented by distinct colored lines. Data illustrates performance differences between models.</alt-text>
</graphic>
</fig>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The tongue segmentation results of different networks on Dataset A. <bold>(A)</bold> represents the original image, <bold>(B)</bold> represents the ground truth, and <bold>(C&#x2013;K)</bold> represent U-Net, DeepLabV3plus, Swin Transformer, SegFormer, SegNeXt, PoolFormer, RTC_TongueNet, TongueSAM, and Ours, respectively. <bold>(L&#x2013;U)</bold> are the local magnification images, using (1) as an example from Dataset A, and show the ground truth, U-Net, DeepLabV3plus, Swin Transformer, SegFormer, SegNeXt, PoolFormer, RTC_TongueNet, TongueSAM, and Ours, respectively.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g006.tif">
<alt-text content-type="machine-generated">Five rows of images display a person's tongue with corresponding red contour maps on a black background. The columns, labeled A to K, show variations in contour shapes for each image. Below, the images L to U depict similar contours with zoomed sections, highlighting detail variations in tongue topography against the black background.</alt-text>
</graphic>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>The tongue segmentation results of different networks on Dataset B. <bold>(A)</bold> represents the original image, <bold>(B)</bold> represents the ground truth, and <bold>(C&#x2013;K)</bold> represent U-Net, DeepLabV3plus, Swin Transformer, SegFormer, SegNeXt, PoolFormer, RTC_TongueNet, TongueSAM, and Ours, respectively. <bold>(L&#x2013;U)</bold> are the local magnification images, using (2) as an example from Dataset B, and show the ground truth, U-Net, DeepLabV3plus, Swin Transformer, SegFormer, SegNeXt, PoolFormer, RTC_TongueNet, TongueSAM, and Ours, respectively.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g007.tif">
<alt-text content-type="machine-generated">Medical images showing a series of tongue appearances in Dataset B, with five photographs of tongues in column (1) to (5) and corresponding analyzed contours in columns A to K. The bottom section expands analysis L to U, detailing specific areas with red contour lines on black backgrounds.</alt-text>
</graphic>
</fig>
<sec id="s4-2-1">
<title>4.2.1 Performance analysis</title>
<p>The model size and inference speed of GA-TongueNet are moderate, as detailed in <xref ref-type="table" rid="T1">Table 1</xref>. Its design strikes a balance, positioning it as neither particularly lightweight nor excessively resource-intensive. Among the nine evaluated models, GA-TongueNet demonstrated competitive performance, surpassing the two models specifically designed for tongue image segmentation. Notably, it achieved the optimal performance in tongue segmentation, underscoring its superior effectiveness in this specialized task.</p>
<p>On Dataset A, GA-TongueNet achieved a Dice of 0.9906 and an IoU of 0.9814, outperforming widely used CNN-based models such as DeepLabV3plus (Dice: 0.9851, IoU: 0.9706), U-Net (Dice: 0.9765, IoU: 0.9541), and SegNeXt (Dice: 0.9890, IoU: 0.9781). It also demonstrated advantages over Transformer-based models, including Swin Transformer (Dice: 0.9863, IoU: 0.9729), SegFormer (Dice: 0.9734, IoU: 0.9481), and PoolFormer (Dice: 0.9845, IoU: 0.9694). While GA-TongueNet&#x2019;s Precision (0.9894) was slightly lower than SegFormer&#x2019;s (0.9909), its Dice (0.9906), and IoU (0.9814) Recall (0.9918) were the highest among all evaluated models. Furthermore, compared to RTC_TongueNet (Dice: 0.9640, IoU: 0.9310) and TongueSAM (Dice: 0.9476, IoU: 0.9005), both of which are specifically designed for tongue image segmentation, GA-TongueNet demonstrated superior performance across all evaluation metrics. As illustrated in <xref ref-type="fig" rid="F5">Figure 5A</xref>, while the performances of the compared models are generally comparable, GA-TongueNet exhibits a more outward trajectory on the radar chart, reflecting its relatively superior overall performance in this task.</p>
<p>On the more challenging and diverse Dataset B, GA-TongueNet demonstrates clear advantages. As shown in <xref ref-type="fig" rid="F5">Figure 5B</xref>, the overall performance of GA-TongueNet is notably superior among the nine evaluated models. Particularly for metrics such as Dice and IoU, GA-TongueNet exhibits a more pronounced outward trajectory, reflecting its relatively superior performance. The model achieved a Dice of 0.9553, IoU of 0.9163, Precision of 0.9501, and Recall of 0.9632, outperforming all competing models. For instance, compared to DeepLabV3plus (Dice: 0.8953, IoU: 0.8334), GA-TongueNet shows improvements of 6.70% in Dice and 9.88% in IoU. Similarly, its IoU exceeds that of U-Net (IoU: 0.7092) and Swin Transformer (IoU: 0.7874) by 20.71% and 12.89%, respectively, highlighting its capability to generalize effectively to complex real-world data. Even when compared to the strong Transformer-based competitor SegFormer, GA-TongueNet achieves better results, with Dice, IoU, Precision, and Recall being 0.0214, 0.0379, 0.0212, and 0.0209 higher, respectively. These results underscore the robustness and adaptability of GA-TongueNet across varying data distributions. Although SegNeXt performed commendably on Dataset A, it encountered significant challenges in generalizing to the complex conditions of Dataset B, achieving an IoU of only 0.7895, substantially lower than GA-TongueNet. Similarly, while PoolFormer achieved a relatively high IoU of 0.8848, it remained 3.16% lower than that of GA-TongueNet. The two models specifically designed for tongue image segmentation, RTC_TongueNet and TongueSAM, also lagged behind GA-TongueNet in comprehensive performance. Notably, RTC_TongueNet exhibited relatively poor generalization ability. These findings highlight the limitations of traditional CNN-based architectures and some Transformer-based designs in addressing the complexities of diverse and challenging scenarios, while reinforcing the stable generalization and adaptability of GA-TongueNet.</p>
<p>Qualitatively, as illustrated in <xref ref-type="fig" rid="F6">Figure 6</xref>, GA-TongueNet demonstrates high-precision boundary delineation on Dataset A. The detailed segmentation performance of each model is further highlighted in <xref ref-type="fig" rid="F6">Figures 6L&#x2013;U</xref>, where local details are examined. While most models exhibit segmentation results that align well with the ground truth, the performance of RTC_TongueNet and TongueSAM shows room for improvement, particularly given the constraints of the current small-scale dataset. In contrast, <xref ref-type="fig" rid="F7">Figure 7</xref> reveals GA-TongueNet&#x2019;s robustness in handling challenging conditions such as varying lighting and complex backgrounds, scenarios that prove difficult for other models. From <xref ref-type="fig" rid="F7">Figures 7L&#x2013;U</xref>, it becomes evident that under uneven illumination, U-Net and DeepLabV3plus struggle with issues of false detection and incomplete region segmentation. Swin Transformer performs admirably in standard acquisition environments but fails to generalize effectively to Dataset B. SegFormer, despite being competitive, encounters challenges such as boundary recognition errors. Similarly, while SegNeXt and PoolFormer exhibit strong performance, they remain slightly inferior to GA-TongueNet in terms of accuracy and consistency. For the tongue-specific models, RTC_TongueNet displays limited generalization ability, making precise segmentation in complex environments challenging. TongueSAM performs relatively better, achieving successful segmentation for most tongue bodies; however, it also exhibits a higher rate of false positives. These comprehensive results underscore GA-TongueNet&#x2019;s ability to achieve accurate and reliable tongue segmentation while maintaining adaptability across diverse and complex data environments. Its superior generalization and robustness further highlight its potential as a dependable tool for tongue image segmentation in real-world applications.</p>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Comprehensive analysis for methods comparison</title>
<p>Under the constraints of small-sample datasets, the experimental results demonstrate that GA-TongueNet achieves remarkable segmentation performance, across both standard and challenging scenarios. Its comprehensive performance surpasses that of current comparison models, highlighting its strong generalization capability and robustness. GA-TongueNet&#x2019;s ability to handle complex backgrounds and diverse conditions effectively makes it particularly well-suited for applications involving limited training data. This adaptability underscores its potential for reliable deployment in both standard acquisition environments and more complex, real-world scenarios.</p>
</sec>
</sec>
<sec id="s4-3">
<title>4.3 Ablation study</title>
<p>To thoroughly investigate the contributions of the proposed components to the overall performance of the model, we conducted an ablation study. Specifically, we compared the SegFormer baseline model (Baseline), the model enhanced with the MDi module (&#x2b;MDi), the model integrated with the DiFP module (&#x2b;DiFP), and the model incorporating both MDi and DiFP modules (Full Model). The experiments were conducted on both Dataset A and Dataset B, using Dice, IoU, Precision, and Recall as evaluation metrics to rigorously assess the effectiveness of the improvement strategies. The results are detailed in <xref ref-type="table" rid="T3">Table 3</xref>, and visually represented in <xref ref-type="fig" rid="F8">Figure 8</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Evaluation metrics data for ablation study.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Dataset</th>
<th align="left">Network</th>
<th align="left">Dice</th>
<th align="left">IoU</th>
<th align="left">Precision</th>
<th align="left">Recall</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="left">Dataset A</td>
<td align="left">Baseline</td>
<td align="left">0.9734</td>
<td align="left">0.9481</td>
<td align="left">0.9909</td>
<td align="left">0.9565</td>
</tr>
<tr>
<td align="left">&#x2b;MDi</td>
<td align="left">0.9863</td>
<td align="left">0.9730</td>
<td align="left">0.9878</td>
<td align="left">0.9848</td>
</tr>
<tr>
<td align="left">&#x2b;DiFP</td>
<td align="left">0.9872</td>
<td align="left">0.9748</td>
<td align="left">
<bold>0.9916</bold>
</td>
<td align="left">0.9829</td>
</tr>
<tr>
<td align="left">Full Model</td>
<td align="left">
<bold>0.9906</bold>
</td>
<td align="left">
<bold>0.9814</bold>
</td>
<td align="left">0.9894</td>
<td align="left">
<bold>0.9918</bold>
</td>
</tr>
<tr>
<td rowspan="4" align="left">Dataset B</td>
<td align="left">Baseline</td>
<td align="left">0.9339</td>
<td align="left">0.8784</td>
<td align="left">0.9289</td>
<td align="left">0.9423</td>
</tr>
<tr>
<td align="left">&#x2b;MDi</td>
<td align="left">0.8213</td>
<td align="left">0.7318</td>
<td align="left">0.7832</td>
<td align="left">0.9135</td>
</tr>
<tr>
<td align="left">&#x2b;DiFP</td>
<td align="left">0.9423</td>
<td align="left">0.8930</td>
<td align="left">0.9453</td>
<td align="left">0.9425</td>
</tr>
<tr>
<td align="left">Full Model</td>
<td align="left">
<bold>0.9553</bold>
</td>
<td align="left">
<bold>0.9163</bold>
</td>
<td align="left">
<bold>0.9501</bold>
</td>
<td align="left">
<bold>0.9632</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values represent the optimal values.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Bar chart of evaluation metrics for ablation study on datasets. <bold>(A)</bold> represents Dataset A and <bold>(B)</bold> represents Dataset B.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g008.tif">
<alt-text content-type="machine-generated">Bar graphs comparing different models' performance across metrics. Graph A shows the Full Model in red outperforming others in Dice, IoU, Precision, and Recall scores, with values above 0.98. Graph B depicts similar metrics with Full Model scores slightly lower, but still leading with scores like 0.9553 in Dice. Models include Baseline, +MDi, +DiFP, and Full Model, with color codes for each.</alt-text>
</graphic>
</fig>
<sec id="s4-3-1">
<title>4.3.1 Performance analysis</title>
<p>The introduction of the MDi and DiFP modules has significantly influenced model performance on both Dataset A and Dataset B, demonstrating complementary effects, as shown in <xref ref-type="table" rid="T3">Table 3</xref>; <xref ref-type="fig" rid="F9">Figure 9</xref>. On Dataset A, the addition of the MDi module improves the Dice and IoU to 0.9863 and 0.9730, respectively, compared to the Baseline model (Dice: 0.9734, IoU: 0.9481). The DiFP module further enhances the performance, achieving a Dice of 0.9872 and an IoU of 0.9748. When both modules are integrated in the Full Model, the performance reaches its peak, with a Dice of 0.9906 and an IoU of 0.9814. These results highlight the ability of the DiFP module to capture detailed structural features and the role of the MDi module in maintaining feature resolution across varying tongue sizes. Notably, as illustrated in <xref ref-type="fig" rid="F9">Figures 9M&#x2013;Q</xref>, the segmentation results of the Full Model align more closely with the ground truth, reducing errors observed in the Baseline and single-module configurations. In particular, <xref ref-type="fig" rid="F9">Figures 9N&#x2013;P</xref> show false positives.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>The tongue segmentation results of the ablation study on datasets. <bold>(A&#x2013;F)</bold> are from Dataset A. <bold>(A)</bold> represents the original image, <bold>(B)</bold> represents the ground truth, and <bold>(C&#x2013;F)</bold> represent Baseline, &#x2b;MDi, &#x2b;DiFP, and Full Model, respectively. <bold>(G&#x2013;L)</bold> are from Dataset B. <bold>(G)</bold> represents the original image, <bold>(H)</bold> represents the ground truth, and <bold>(I&#x2013;L)</bold> represent Baseline, &#x2b;MDi, &#x2b;DiFP, and Full Model, respectively. <bold>(M&#x2013;Q)</bold> are the local magnification images from Dataset A, using (4) as an example, and show the ground truth, Baseline, &#x2b;MDi, &#x2b;DiFP, and Full Model, respectively. <bold>(R&#x2013;V)</bold> are the local magnification images from Dataset B, using (8) as an example, and show the ground truth, Baseline, &#x2b;MDi, &#x2b;DiFP, and Full Model, respectively.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g009.tif">
<alt-text content-type="machine-generated">Two datasets labeled A and B compare tongue images with their corresponding red-outlined segmentation masks on black backgrounds. Rows 1 to 5 feature Dataset A, while rows 6 to 10 feature Dataset B. The first column shows the original tongue images, followed by columns of their segmented versions. The bottom section highlights detailed views, focusing on different segmentation aspects for specific images.</alt-text>
</graphic>
</fig>
<p>On Dataset B, the performance trend reflects different behaviors under more complex conditions. The Baseline model achieves satisfactory results, with a Dice of 0.9339 and an IoU of 0.8784. However, the inclusion of the MDi module unexpectedly leads to a decrease in performance, with Dice and IoU dropping to 0.8213 and 0.7318, respectively. This decline suggests that the MDi module introduces instability in handling diverse and challenging scenarios. In contrast, the &#x2b;DiFP model performs better, achieving a Dice of 0.9423 and an IoU of 0.8930, which surpasses the Baseline performance. When MDi and DiFP are integrated, the Full Model demonstrates the best generalization capability, achieving a Dice of 0.9553, an IoU of 0.9163, a Precision of 0.9501, and a Recall of 0.9632. These results underscore the complementary effects of the two modules in improving segmentation performance under challenging conditions. <xref ref-type="fig" rid="F9">Figures 9R&#x2013;V</xref> further corroborates these findings. In Dataset B&#x2019;s challenging scenarios, <xref ref-type="fig" rid="F9">Figures 9S, T</xref> show false positives. The Full Model achieves more accurate segmentation of the tongue region, minimizing both false positives and missed areas.</p>
</sec>
<sec id="s4-3-2">
<title>4.3.2 Comprehensive analysis for ablation study</title>
<p>The ablation study reveals the complementary contributions of the MDi and DiFP modules to the overall model performance. Dataset A, serving as the training set and characterized by relatively standardized data, shows consistent improvements when either module is added individually. The Full Model, which integrates both MDi and DiFP, achieves the highest scores across all evaluation metrics, indicating effective synergy between these components.</p>
<p>In contrast, Dataset B, used exclusively as a testing set and representing more complex and diverse scenarios, exhibits different behavior. The MDi module alone results in a drop in performance compared to the Baseline, with the IoU decreasing from 0.8784 to 0.7318. This suggests that MDi, when applied independently, may introduce instability or reduced robustness in challenging environments. The DiFP module performs more reliably on Dataset B and improves several metrics over the Baseline, though it does not fully surpass it on all measures. Importantly, the combined Full Model leverages the complementary strengths of MDi and DiFP to achieve superior performance, demonstrating better generalization and robustness on Dataset B despite its complexity.</p>
<p>These results indicate that while the MDi module may have limitations when deployed independently on unseen complex data, its integration with the DiFP module provides a more balanced and stable architecture. The synergy between these modules enhances the model&#x2019;s ability to generalize from training on Dataset A to challenging test scenarios in Dataset B, thereby improving segmentation accuracy and robustness in practical applications.</p>
</sec>
</sec>
<sec id="s4-4">
<title>4.4 Generalization verification</title>
<p>To verify the generalization ability of GA-TongueNet, we first used the 95% confidence interval (CI) and 5-fold cross-validation. Secondly, we took the MAE with excellent generalization ability as the backbone of our model to further verify the generalization ability.</p>
<sec id="s4-4-1">
<title>4.4.1 Preliminary verification</title>
<p>Comparison experiments and ablation studies indicate that GA-TongueNet, when trained on the standard environment of Dataset A, achieves superior performance in tongue segmentation when applied to the more complex and natural scenarios of Dataset B. Preliminary validation of the generalization ability by up to 5-fold cross-validation trained yielded stable and good metrics: Dice (98.95 <inline-formula id="inf35">
<mml:math id="m42">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.06%), IoU (97.92 <inline-formula id="inf36">
<mml:math id="m43">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.12%), Precision (98.88 <inline-formula id="inf37">
<mml:math id="m44">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.19%), and Recall (99.08 <inline-formula id="inf38">
<mml:math id="m45">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.15%). These results reflect the robustness and consistency of the model, and to a certain extent, rule out the possibility of overfitting. In addition, in terms of CI, we first calculated the individual metrics (e.g., IoU, Dice, Precision, Recall) for each of the 30 images in Dataset A and 100 images in Dataset B. The mean and variance of these metrics were then determined, providing the basis for deriving the final CI. From the experimental results, Dataset A performs well as Dice 98.68% (95% CI [97.18%, 97.62%]), IoU 97.40% (95% CI [98.57%, 98.79%]), Precision 99.35% (95% CI [99.11%, 99.58%]) and Recall 98.04% (95% CI [97.78%, 98.29%]). In comparison, while the performance on Dataset B shows slight degradation, the metrics remain robust: Dice 95.53% (95% CI [94.85%, 96.21%]), IoU 91.63% (95% CI [90.46%, 92.81%]), Precision 95.01% (95% CI [94.03%, 95.99%]) and Recall 96.32% (95% CI [95.48%, 97.16%]). As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, Dataset B is extremely different from Dataset A in terms of lighting, background, and shooting angle, which somewhat validates the model&#x2019;s potential for real-world application in unseen environments.</p>
</sec>
<sec id="s4-4-2">
<title>4.4.2 Further verification</title>
<p>The MAE framework has been shown to enhance model generalization through mechanisms such as unsupervised learning, high-ratio masking, and latent representation learning (<xref ref-type="bibr" rid="B10">He et al., 2022</xref>). While the comparative and ablation experiments presented earlier effectively demonstrate the generalization ability of the original GA-TongueNet structure (the proposed model in this study), we conducted further verification by integrating MAE as the backbone. Specifically, MAE was pre-trained on a dataset of 4,213 tongue images, 70% of which were augmented versions of Dataset A (using techniques such as rotation and color transformation), while the remaining images resembled those in Dataset B.</p>
<p>Following this pre-training, we applied two strategies&#x2014;Freezing and Non-Freezing&#x2014;for training GA-TongueNet with MAE as the backbone. These experiments were designed to strengthen the evidence supporting the generalization capability of the original GA-TongueNet structure. The experimental results for these models, including the original GA-TongueNet, and the MAE-based variants trained using Freezing and Non-Freezing strategies, are presented in <xref ref-type="table" rid="T4">Table 4</xref>; <xref ref-type="fig" rid="F10">Figures 10</xref>, <xref ref-type="fig" rid="F11">11</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Evaluation metrics data for generalization verification.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Dataset</th>
<th align="left">Network</th>
<th align="left">Dice</th>
<th align="left">IoU</th>
<th align="left">Precision</th>
<th align="left">Recall</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="left">Dataset A</td>
<td align="left">Freezing</td>
<td align="left">0.9843</td>
<td align="left">0.9690</td>
<td align="left">0.9866</td>
<td align="left">0.9820</td>
</tr>
<tr>
<td align="left">Non-Freezing</td>
<td align="left">0.9857</td>
<td align="left">0.9718</td>
<td align="left">0.9874</td>
<td align="left">0.9840</td>
</tr>
<tr>
<td align="left">Ours</td>
<td align="left">
<bold>0.9906</bold>
</td>
<td align="left">
<bold>0.9814</bold>
</td>
<td align="left">
<bold>0.9894</bold>
</td>
<td align="left">
<bold>0.9918</bold>
</td>
</tr>
<tr>
<td rowspan="3" align="left">Dataset B</td>
<td align="left">Freezing</td>
<td align="left">0.7488</td>
<td align="left">0.6207</td>
<td align="left">0.6867</td>
<td align="left">0.8818</td>
</tr>
<tr>
<td align="left">Non-Freezing</td>
<td align="left">0.8715</td>
<td align="left">0.7955</td>
<td align="left">0.8179</td>
<td align="left">
<bold>0.9704</bold>
</td>
</tr>
<tr>
<td align="left">Ours</td>
<td align="left">
<bold>0.9553</bold>
</td>
<td align="left">
<bold>0.9163</bold>
</td>
<td align="left">
<bold>0.9501</bold>
</td>
<td align="left">0.9632</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values represent the optimal values.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Bar chart of evaluation metrics for generalization verification on datasets. <bold>(A)</bold> represents Dataset A and <bold>(B)</bold> represents Dataset B.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g010.tif">
<alt-text content-type="machine-generated">Bar charts labeled A and B compare scores of different methods: &#x22;Freezing&#x22; in green, &#x22;Non-freezing&#x22; in blue, and &#x22;Ours&#x22; in red. Metrics include Dice, IoU, Precision, and Recall. &#x22;Ours&#x22; consistently scores higher across all metrics.</alt-text>
</graphic>
</fig>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>The tongue segmentation results of generalization verification on datasets. <bold>(A&#x2013;E)</bold> are from Dataset A. <bold>(A)</bold> represents the original image, <bold>(B)</bold> represents the ground truth, and <bold>(C&#x2013;E)</bold> represent Freezing, Non-freezing, and Ours, respectively. <bold>(F&#x2013;J)</bold> are from Dataset B. <bold>(F)</bold> represents the original image, <bold>(G)</bold> represents the ground truth, and <bold>(H&#x2013;J)</bold> represent Freezing, Non-freezing, and Ours, respectively. <bold>(K&#x2013;N)</bold> are the local magnification images from Dataset A, using (1) as an example, and show the ground truth, Freezing, Non-freezing, and Ours, respectively. <bold>(O&#x2013;R)</bold> are the local magnification images from Dataset B, using (6) as an example, and show the ground truth, Freezing, Non-freezing, and Ours, respectively.</p>
</caption>
<graphic xlink:href="fphys-16-1617647-g011.tif">
<alt-text content-type="machine-generated">Comparison of tongue images from two datasets, A and B. Dataset A (images 1-5) shows tongues with edge detection outlines in red. Dataset B (images 6-10) shows similar tongue images with varied outlines. Enlarged sections (K-R) highlight specific edge details, focusing on texture differences in datasets.</alt-text>
</graphic>
</fig>
<p>The evaluation metrics for Dataset A and Dataset B are summarized in <xref ref-type="table" rid="T4">Table 4</xref>. For Dataset A, the proposed original GA-TongueNet shows consistent improvements over the MAE-based GA-TongueNet under both Freezing and Non-Freezing strategies. Across all metrics, including Dice, IoU, Precision, and Recall, the original GA-TongueNet achieves the highest scores, with Dice of 0.9906, IoU of 0.9814, Precision of 0.9894, and Recall of 0.9918. These results suggest that the architecture is well-suited for handling the relatively controlled conditions in Dataset A.</p>
<p>For Dataset B, the Non-Freezing strategy outperforms the Freezing strategy, achieving a Dice score of 0.8715, IoU of 0.7955, Precision of 0.8179, and Recall of 0.9704. This indicates that allowing parameter adjustments during training can help the model adapt better to the diverse and complex scenarios in Dataset B. Nonetheless, the performance of the Non-Freezing strategy remains below that of the original GA-TongueNet. The proposed original GA-TongueNet achieves Dice, IoU, Precision, and Recall scores of 0.9553, 0.9163, 0.9501, and 0.9632, respectively, on Dataset B. Compared to the Non-Freezing strategy, these values represent noticeable improvements in Dice (9.6%), IoU (15.1%), and Precision (13.2%), while Recall is marginally lower by 0.7%. These findings suggest that the original GA-TongueNet is capable of addressing complex scenarios in Dataset B effectively, while the slight trade-off in Recall indicates potential areas for further refinement. Overall, the results indicate that the proposed model achieves balanced performance across datasets with varying complexity, demonstrating a degree of robustness and adaptability.</p>
<p>
<xref ref-type="fig" rid="F11">Figure 11</xref> provides a visual comparison of the segmentation results across the two datasets. In Dataset A, while both the Freezing and Non-Freezing strategies demonstrate reasonable performance, as shown in <xref ref-type="fig" rid="F11">Figures 11L,M</xref>, challenges remain in handling edge details, with occasional false positives observed. For Dataset B, the performance of both strategies diminishes in the presence of complex scenarios, such as foreign objects like tongue studs. As illustrated in <xref ref-type="fig" rid="F11">Figures 11P,Q</xref>, these approaches face difficulties in effectively addressing edge detail segmentation under such conditions. In contrast, original GA-TongueNet demonstrates improved robustness, providing more accurate segmentation of the tongue body while minimizing significant false positives or omissions. These observations highlight GA-TongueNet&#x2019;s potential for enhanced performance in varied and challenging environments, though opportunities for further improvement remain.</p>
</sec>
<sec id="s4-4-3">
<title>4.4.3 Comprehensive analysis for generalization verification</title>
<p>The experimental results indicate that in preliminary evaluations, the proposed GA-TongueNet demonstrated favorable performance, providing an initial validation of its generalization ability. Further testing revealed that, compared to its MAE-backbone variant, the original GA-TongueNet exhibited superior generalization and robustness. On Dataset A, characterized by relatively standard conditions, the original GA-TongueNet leveraged its architectural design to achieve high segmentation performance. On Dataset B, which features more complex backgrounds and diverse variations, the model demonstrated stronger adaptability by effectively mitigating interference and maintaining consistent segmentation quality. The performance gap between the original GA-TongueNet and the MAE-backbone variant was particularly pronounced in challenging scenarios, underscoring the potential advantages of the proposed architecture. These findings support the conclusion that the original GA-TongueNet possesses commendable generalization capabilities. By effectively addressing diverse and complex conditions, including unseen environments, the original GA-TongueNet demonstrates promise as a reliable tongue image segmentation model for practical applications.</p>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>In TCM, tongue features are considered key indicators of an individual&#x2019;s physiological functions and pathological changes. However, in computer-aided diagnosis, the accuracy of tongue diagnosis can be significantly affected by confounding factors such as teeth, facial regions, and other background elements. Precise tongue image segmentation is therefore critical for enhancing diagnostic accuracy.</p>
<p>From a clinical perspective, accurate tongue image segmentation is fundamental to improving the performance of computer-assisted tongue diagnosis systems, particularly in mobile applications. As illustrated in Dataset B of <xref ref-type="fig" rid="F1">Figure 1</xref>, most tongue images captured in real-world scenarios often include complex backgrounds. Without proper segmentation, the surrounding environment of the tongue body introduces substantial noise, which can compromise the accuracy of the analysis. Segmentation isolates the tongue body, allowing diagnostic models to focus exclusively on relevant features without being influenced by extraneous factors. This targeted approach significantly enhances the precision and reliability of tongue diagnosis.</p>
<p>Tongue segmentation is crucial for extracting disease-related features, such as the color of the tongue body and the distribution and thickness of the tongue coating. According to TCM theory, these features are strongly correlated with the functional states of internal organs. For instance, variations in the thickness and color of the tongue coating may reflect digestive system abnormalities or indicate internal dampness or heat, providing valuable insights for syndrome differentiation and treatment planning. Additionally, region-specific analysis of the tongue&#x2014;such as the tip, center, and base&#x2014;enables a nuanced understanding of organ-specific functions. For example, a red or yellow coating on the tongue tip may suggest hyperactivity of heart fire, while a greasy coating on the tongue base could indicate renal insufficiency. These correlations underscore the diagnostic value of precise tongue segmentation. The robustness of segmentation algorithms is equally critical for their application in diverse clinical environments. Variations in lighting conditions, background interference, and differences in tongue posture can introduce significant variability in tongue images. A well-designed segmentation model capable of addressing these challenges ensures consistent and accurate image analysis, irrespective of the imaging environment. This consistency is vital for generating reliable diagnostic data across populations and regions. By overcoming these challenges, tongue segmentation significantly enhances the clinical utility of computer-assisted tongue diagnosis, enabling applications such as early disease detection, comprehensive health status evaluation, and effective monitoring of treatment outcomes.</p>
<p>With the rapid advancement of deep learning techniques, CNN, renowned for their robust feature extraction capabilities, have been widely applied in tongue segmentation. Notable models include OET-NET (<xref ref-type="bibr" rid="B13">Huang et al., 2022</xref>), QA-TSN (<xref ref-type="bibr" rid="B14">Jia et al., 2025</xref>), LAIU-Net (<xref ref-type="bibr" rid="B22">Marhamati et al., 2023</xref>), and HPA-UNet (<xref ref-type="bibr" rid="B34">Yao et al., 2024</xref>). In addition, models leveraging the self-attention mechanism of Transformers, such as PriTongueNet (<xref ref-type="bibr" rid="B12">Huang et al., 2025</xref>) and Tongue-LiteSAM (<xref ref-type="bibr" rid="B27">Tan et al., 2025</xref>), have demonstrated promising segmentation performance. However, these models often require extensive training datasets and exhibit limited generalization ability.</p>
<p>In comparison, our proposed GA-TongueNet demonstrates high segmentation accuracy despite relying on a relatively modest training dataset. It utilizes the publicly available Dataset A (<xref ref-type="bibr" rid="B2">BioHit, 2014</xref>), consisting of 300 tongue images, and an additional 100 images collected from diverse and complex contexts for generalization testing. Comparative experiments against four representative models&#x2014;spanning both CNN and Transformer architectures&#x2014;demonstrate that GA-TongueNet consistently outperforms these models. It effectively addresses the challenges associated with limited and heterogeneous datasets, showcasing its potential to alleviate existing limitations in tongue image segmentation tasks.</p>
<p>To further explore the generalization potential of GA-TongueNet, we integrated a pre-trained MAE (<xref ref-type="bibr" rid="B10">He et al., 2022</xref>) as its backbone. MAE, known for its proficiency in unsupervised learning, high-rate masking, and latent representation learning, was trained on a large-scale tongue image dataset. However, experimental results reveal that this modified architecture underperforms relative to the original GA-TongueNet. This unexpected outcome underscores the inherent adaptability and robustness of the original GA-TongueNet design, which maintains high segmentation accuracy even for tongue images outside the training set.</p>
<p>Although the results are encouraging, there are still some deficiencies in this study. Judging from the experimental results, GA-TongueNet may make false positives when facing unfamiliar situations, such as foreign objects on the tongue (such as punctured tongue nails, tongue perforations), and mistakenly think it is the tongue body. For areas with poor lighting, such as the back of the tongue, false positives may occur. Furthermore, in the fine segmentation of the tongue edge, the result is not always optimal. This is more obvious at the boundary between the tongue and adjacent structures such as teeth or lips, where the division may appear blurred and less precise. As for the model itself, although the current model has achieved the best segmentation effect in experiments, the computational cost is not the lowest. It requires higher computing resources and a longer time. How to reduce computing costs and computing resources has become one of the directions we are actively improving, hoping to be deployed on more devices. Furthermore, the MDi module and the DiFP module need to work in synergy to achieve the optimal performance.</p>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>In this article, we have presented GA-TongueNet, a model designed for tongue image segmentation, addressing the challenges of semantic segmentation in complex environments. By incorporating the DiFP and MDi modules, the model demonstrates the ability to achieve multi-scale feature fusion and effectively capture contextual information. Despite being trained only on tongue images obtained in standard acquisition environments, GA-TongueNet shows promising performance in segmenting images captured under challenging lighting conditions.The experimental results suggest that GA-TongueNet performs well in terms of segmentation accuracy and generalization across diverse and complex environments. While there is still potential for improvement, these findings indicate that GA-TongueNet could serve as a useful approach for tongue image segmentation in real-world applications.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://github.com/BioHit/TongeImageDataset">https://github.com/BioHit/TongeImageDataset</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>ZD: Methodology, Conceptualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. LZ: Methodology, Conceptualization, Funding acquisition, Writing &#x2013; review and editing. YF: Formal Analysis, Writing &#x2013; review and editing. HM: Writing &#x2013; review and editing, Data curation. CS: Writing &#x2013; original draft, Methodology. YZ: Writing &#x2013; original draft, Investigation. PL: Writing &#x2013; review and editing, Project administration.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was funded by the Key R &#x26; D and Promotion Projects in Henan Province under Grant 242102240117, Henan Provincial Education Science Planning 2024 General Project under Grant 2024YB0077, Natural Science Project of Zhengzhou Science and Technology Bureau under Grant 22ZZRDZX43, Natural Science Foundation of Henan under Grant 242300421709 and 252300420366, Open Project of Institute for Complexity Science of Henan University of Technology under Grant CSKFJJ-2024-21, Henan University of Technology Undergraduate Innovation and Entrepreneurship Training Project under Grant PX-38244814, the Research and Practice Project on Teaching Reform in Higher Education of Henan Province under Grant 2024SJGLX0095, and Cultivation Project of National Natural Science Foundation of Henan University of Technology 2024PYJH034.</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ahn</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>A spatial guided self-supervised clustering network for medical image segmentation</article-title>,&#x201d; in <source>Medical image computing and computer assisted intervention &#x2013; miccai 2021</source>. Editors <person-group person-group-type="editor">
<name>
<surname>de Bruijne</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cattin</surname>
<given-names>P. C.</given-names>
</name>
<name>
<surname>Cotin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Padoy</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Speidel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>379</fpage>&#x2013;<lpage>388</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<collab>BioHit</collab> (<year>2014</year>). <article-title>Tonge image dataset</article-title>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Tsrnet: tongue image segmentation with global and local refinement</article-title>. <source>Displays</source> <volume>81</volume>, <fpage>102601</fpage>. <pub-id pub-id-type="doi">10.1016/j.displa.2023.102601</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Tonguesam: an universal tongue segmentation model based on sam with zero-shot</article-title>,&#x201d; in <source>2023 IEEE international conference on bioinformatics and biomedicine (BIBM)</source>, <fpage>4520</fpage>&#x2013;<lpage>4526</lpage>. <pub-id pub-id-type="doi">10.1109/BIBM58861.2023.10385570</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L.-C.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Papandreou</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Schroff</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Adam</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Encoder-decoder with atrous separable convolution for semantic image segmentation</article-title>,&#x201d; in <source>Computer vision &#x2013; eccv 2018</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Ferrari</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Hebert</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sminchisescu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Weiss</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>833</fpage>&#x2013;<lpage>851</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Improving ultrasound tongue contour extraction using u-net and shape consistency-based regularizer</article-title>,&#x201d; in <source>Icassp 2021 - 2021 IEEE international conference on acoustics, speech and signal processing (ICASSP)</source>, <fpage>6443</fpage>&#x2013;<lpage>6447</lpage>. <pub-id pub-id-type="doi">10.1109/ICASSP39728.2021.9414420</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Unsupervised representation learning for tissue segmentation in histopathological images: from global to local contrast</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>41</volume>, <fpage>3611</fpage>&#x2013;<lpage>3623</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2022.3191398</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Adaptive active contour model based automatic tongue image segmentation</article-title>,&#x201d; in <source>2016 9th international congress on image and signal processing, BioMedical engineering and informatics</source> (<publisher-name>CISP-BMEI</publisher-name>), <fpage>1386</fpage>&#x2013;<lpage>1390</lpage>. <pub-id pub-id-type="doi">10.1109/CISP-BMEI.2016.7852933</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>M.-H.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>C.-Z.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>M.-M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>S.-m.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Segnext: rethinking convolutional attention design for semantic segmentation</article-title>,&#x201d;. <source>Advances in neural information processing systems</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Koyejo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mohamed</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Agarwal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Belgrave</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Oh</surname>
<given-names>A.</given-names>
</name>
</person-group> (<publisher-name>Curran Associates, Inc.</publisher-name>), <volume>35</volume>, <fpage>1140</fpage>&#x2013;<lpage>1156</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Masked autoencoders are scalable vision learners</article-title>,&#x201d; in <source>2022 IEEE/CVF conference on computer vision and pattern recognition (CVPR)</source>, <fpage>15979</fpage>&#x2013;<lpage>15988</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01553</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>M.-C.</given-names>
</name>
<name>
<surname>Lan</surname>
<given-names>K.-C.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>W.-C.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.-C.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>T.-J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>C.-P.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Automated tongue diagnosis on the smartphone and its applications</article-title>. <source>Comput. Methods Programs Biomed.</source> <volume>174</volume>, <fpage>51</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmpb.2017.12.029</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Attention guided tongue segmentation with geometric knowledge in complex environments</article-title>. <source>Biomed. Signal Process. Control</source> <volume>104</volume>, <fpage>107426</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2024.107426</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>A novel tongue segmentation method based on improved u-net</article-title>. <source>Neurocomputing</source> <volume>500</volume>, <fpage>73</fpage>&#x2013;<lpage>89</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2022.05.023</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jia</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Fei</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Qa-tsn: quickaccurate tongue segmentation net</article-title>. <source>Knowledge-Based Syst.</source> <volume>307</volume>, <fpage>112648</fpage>. <pub-id pub-id-type="doi">10.1016/j.knosys.2024.112648</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Le</surname>
<given-names>N. Q. K.</given-names>
</name>
<name>
<surname>Yapp</surname>
<given-names>E. K. Y.</given-names>
</name>
<name>
<surname>Nagasundaram</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Chua</surname>
<given-names>M. C. H.</given-names>
</name>
<name>
<surname>Yeh</surname>
<given-names>H.-Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Computational identification of vesicular transport proteins from sequences using deep gated recurrent units architecture</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>17</volume>, <fpage>1245</fpage>&#x2013;<lpage>1254</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2019.09.005</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tu</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Establishment of noninvasive diabetes risk prediction model based on tongue features and machine learning techniques</article-title>. <source>Int. J. Med. Inf.</source> <volume>149</volume>, <fpage>104429</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2021.104429</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A facial geometry based detection model for face manipulation using cnn-lstm architecture</article-title>. <source>Inf. Sci.</source> <volume>633</volume>, <fpage>370</fpage>&#x2013;<lpage>383</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2023.03.079</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>T.-Y.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hariharan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Belongie</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Feature pyramid networks for object detection</article-title>,&#x201d; in <source>2017 IEEE conference on computer vision and pattern recognition (CVPR)</source>, <fpage>936</fpage>&#x2013;<lpage>944</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2017.106</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Application of an improved grab cut method in tongue image segmentation</article-title>,&#x201d; in <source>Intelligent computing methodologies</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Huang</surname>
<given-names>D.-S.</given-names>
</name>
<name>
<surname>Gromiha</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hussain</surname>
<given-names>A.</given-names>
</name>
</person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>484</fpage>&#x2013;<lpage>495</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Automatic lung segmentation in chest x-ray images using improved u-net</article-title>. <source>Sci. Rep.</source> <volume>12</volume>, <fpage>8649</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-12743-y</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Swin transformer: hierarchical vision transformer using shifted windows</article-title>,&#x201d; in <source>2021 IEEE/CVF international conference on computer vision (ICCV)</source>, <fpage>9992</fpage>&#x2013;<lpage>10002</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00986</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marhamati</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Latifi Zadeh</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Mozhdehi Fard</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Arafat Hussain</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jafarnezhad</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jafarnezhad</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Laiu-net: a learning-to-augment incorporated robust u-net for depressed humans&#x2019; tongue segmentation</article-title>. <source>Displays</source> <volume>76</volume>, <fpage>102371</fpage>. <pub-id pub-id-type="doi">10.1016/j.displa.2023.102371</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Monica</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Shreeharsha</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Falkowski-Gilski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Falkowska-Gilska</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Awasthy</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Phadke</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Melanoma skin cancer detection using mask-rcnn with modified gru model</article-title>. <source>Front. Physiology</source> <volume>14</volume>, <fpage>1324042</fpage>&#x2013;<lpage>2023</lpage>. <pub-id pub-id-type="doi">10.3389/fphys.2023.1324042</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Slim unetr: scale hybrid transformers to efficient 3d medical image segmentation under limited computational resources</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>43</volume>, <fpage>994</fpage>&#x2013;<lpage>1005</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2023.3326188</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A novel tongue feature extraction method on mobile devices</article-title>. <source>Biomed. Signal Process. Control</source> <volume>80</volume>, <fpage>104271</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2022.104271</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ronneberger</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Brox</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>U-net: convolutional networks for biomedical image segmentation</article-title>,&#x201d; in <source>Medical image computing and computer-assisted intervention &#x2013; miccai 2015</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Navab</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hornegger</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wells</surname>
<given-names>W. M.</given-names>
</name>
<name>
<surname>Frangi</surname>
<given-names>A. F.</given-names>
</name>
</person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>234</fpage>&#x2013;<lpage>241</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Tongue-litesam: a lightweight model for tongue image segmentation with zero-shot</article-title>. <source>IEEE Access</source> <volume>13</volume>, <fpage>11689</fpage>&#x2013;<lpage>11703</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2025.3528658</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Rtc_tonguenet: an improved tongue image segmentation model based on deeplabv3</article-title>. <source>Digit. Health</source> <volume>10</volume>, <fpage>20552076241242773</fpage>. <pub-id pub-id-type="doi">10.1177/20552076241242773</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tng</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>N. Q. K.</given-names>
</name>
<name>
<surname>Yeh</surname>
<given-names>H.-Y.</given-names>
</name>
<name>
<surname>Chua</surname>
<given-names>M. C. H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Improved prediction model of protein lysine crotonylation sites using bidirectional recurrent neural networks</article-title>. <source>J. Proteome Res.</source> <volume>21</volume>, <fpage>265</fpage>&#x2013;<lpage>273</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jproteome.1c00848</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>A novel framework for tongue feature extraction framework based on sublingual vein segmentation</article-title>. <source>IEEE Trans. NanoBioscience</source>, <fpage>1</fpage>&#x2013;<lpage>1doi</lpage>. <pub-id pub-id-type="doi">10.1109/TNB.2024.3462461</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Artificial intelligence in tongue diagnosis: using deep convolutional neural network for recognizing unhealthy tongue with tooth-mark</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>18</volume>, <fpage>973</fpage>&#x2013;<lpage>980</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2020.04.002</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Anandkumar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Alvarez</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Segformer: simple and efficient design for semantic segmentation with transformers</article-title>,&#x201d;. <source>Advances in neural information processing systems</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Ranzato</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Beygelzimer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dauphin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Vaughan</surname>
<given-names>J. W.</given-names>
</name>
</person-group> (<publisher-name>Curran Associates, Inc.</publisher-name>), <volume>34</volume>, <fpage>12077</fpage>&#x2013;<lpage>12090</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Multi-task joint learning model for segmenting and classifying tongue images using a deep neural network</article-title>. <source>IEEE J. Biomed. Health Inf.</source> <volume>24</volume>, <fpage>2481</fpage>&#x2013;<lpage>2489</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2020.2986376</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shankar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abidi</surname>
<given-names>M. H.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Hpa-unet: a hybrid post-processing attention u-net for tongue segmentation</article-title>. <source>IEEE J. Biomed. Health Inf.</source>, <fpage>1</fpage>&#x2013;<lpage>12doi</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2024.3446623</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L. T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Armstrong</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Deen</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Convolutional neural networks for medical image analysis: state-of-the-art, comparisons, improvement and perspectives</article-title>. <source>Neurocomputing</source> <volume>444</volume>, <fpage>92</fpage>&#x2013;<lpage>110</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2020.04.157</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Si</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). &#x201c;<article-title>Metaformer is actually what you need for vision</article-title>,&#x201d; in <source>2022 IEEE/CVF conference on computer vision and pattern recognition (CVPR)</source>, <fpage>10809</fpage>&#x2013;<lpage>10819</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01055</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>N. Q. K.</given-names>
</name>
<name>
<surname>Chua</surname>
<given-names>M. C. H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Prediction of anticancer peptides based on an ensemble model of deep learning and machine learning using ordinal positional encoding</article-title>. <source>Briefings Bioinforma.</source> <volume>24</volume>, <fpage>bbac630</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbac630</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Multi-task feature selection with sparse regularization to extract common and task-specific features</article-title>. <source>Neurocomputing</source> <volume>340</volume>, <fpage>76</fpage>&#x2013;<lpage>89</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2019.02.035</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Crookes</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Pe-net: a parallel framework for 3d inferior mesenteric artery segmentation</article-title>. <source>Front. Physiology</source> <volume>14</volume>, <fpage>1308987</fpage>&#x2013;<lpage>2023</lpage>. <pub-id pub-id-type="doi">10.3389/fphys.2023.1308987</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Multi-label body constitution recognition via hwmixer-mlp for facial and tongue images</article-title>. <source>Expert Syst. Appl.</source> <volume>269</volume>, <fpage>126383</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2025.126383</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gui</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>N. Q. K.</given-names>
</name>
<name>
<surname>Chua</surname>
<given-names>M. C. H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Improved prediction model of protein and peptide toxicity by integrating channel attention into a convolutional neural network and gated recurrent units</article-title>. <source>ACS Omega</source> <volume>7</volume>, <fpage>40569</fpage>&#x2013;<lpage>40577</lpage>. <pub-id pub-id-type="doi">10.1021/acsomega.2c05881</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lai</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Weakly supervised deep learning for tooth-marked tongue recognition</article-title>. <source>Front. Physiology</source> <volume>13</volume>, <fpage>847267</fpage>&#x2013;<lpage>2022</lpage>. <pub-id pub-id-type="doi">10.3389/fphys.2022.847267</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>