<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioeng. Biotechnol.</journal-id>
<journal-title>Frontiers in Bioengineering and Biotechnology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioeng. Biotechnol.</abbrev-journal-title>
<issn pub-type="epub">2296-4185</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1392513</article-id>
<article-id pub-id-type="doi">10.3389/fbioe.2024.1392513</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioengineering and Biotechnology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Tongue feature recognition to monitor rehabilitation: deep neural network with visual attention mechanism</article-title>
<alt-title alt-title-type="left-running-head">Yi et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbioe.2024.1392513">10.3389/fbioe.2024.1392513</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Yi</surname>
<given-names>Zhengheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lai</surname>
<given-names>Xinsheng</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sun</surname>
<given-names>Aining</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Fang</surname>
<given-names>Senlin</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2667042/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Shenzhen Fuyong People&#x2019;s Hospital</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Guangzhou University of Chinese Medicine</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>National Famous Traditional Chinese Medicine Expert LAI Xin-sheng Inheritance Studio</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Guangdong Zhengyuanchun Traditional Chinese Medicine Clinic Co., Ltd</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Faculty of Data Science</institution>, <institution>City University of Macau</institution>, <addr-line>Macau</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2323933/overview">Wujing Cao</ext-link>, Chinese Academy of Sciences (CAS), China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2673202/overview">Dong Wang</ext-link>, University of Exeter, United Kingdom</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2677007/overview">Jian Cui</ext-link>, Institute of Artificial Intelligence, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Senlin Fang, <email>D22092100360@cityu.edu.mo</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>09</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1392513</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>16</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Yi, Lai, Sun and Fang.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Yi, Lai, Sun and Fang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Objective</title>
<p>We endeavor to develop a novel deep learning architecture tailored specifically for the analysis and classification of tongue features, including color, shape, and coating. Unlike conventional methods based on architectures like VGG or ResNet, our proposed method aims to address the challenges arising from their extensive size, thereby mitigating the overfitting problem. Through this research, we aim to contribute to the advancement of techniques in tongue feature recognition, ultimately leading to more precise diagnoses and better patient rehabilitation in Traditional Chinese Medicine (TCM).</p>
</sec>
<sec>
<title>Methods</title>
<p>In this study, we introduce TGANet (Tongue Feature Attention Network) to enhance model performance. TGANet utilizes the initial five convolutional blocks of pre-trained VGG16 as the backbone and integrates an attention mechanism into this backbone. The integration of the attention mechanism aims to mimic human cognitive attention, emphasizing model weights on pivotal regions of the image. During the learning process, the allocation of attention weights facilitates the interpretation of causal relationships in the model&#x2019;s decision-making.</p>
</sec>
<sec>
<title>Results</title>
<p>Experimental results demonstrate that TGANet outperforms baseline models, including VGG16, ResNet18, and TSC-WNet, in terms of accuracy, precision, F1 score, and AUC metrics. Additionally, TGANet provides a more intuitive and meaningful understanding of tongue feature classification models through the visualization of attention weights.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>In conclusion, TGANet presents an effective approach to tongue feature classification, addressing challenges associated with model size and overfitting. By leveraging the attention mechanism and pre-trained VGG16 backbone, TGANet achieves superior performance metrics and enhances the interpretability of the model&#x2019;s decision-making process. The visualization of attention weights contributes to a more intuitive understanding of the classification process, making TGANet a promising tool in tongue diagnosis and rehabilitation.</p>
</sec>
</abstract>
<kwd-group>
<kwd>traditional Chinese medicine</kwd>
<kwd>tongue feature recognition</kwd>
<kwd>deep neural network</kwd>
<kwd>attention mechanism</kwd>
<kwd>rehabilitation</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Biomechanics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Traditional Chinese Medicine (TCM) practitioners monitor the rehabilitation process by carefully observing and analyzing the patient&#x2019;s tongue. This method not only aids in determining the progression of the illness but also provides crucial clues for rehabilitation <xref ref-type="bibr" rid="B2">Du et al. (2024)</xref>. Tongue diagnosis plays a pivotal role in the rehabilitation process as changes in the tongue can reflect the overall health condition of the patient. By monitoring features such as the color, shape, and moisture of the tongue, TCM practitioners can assess the progress of the patient&#x2019;s rehabilitation and adjust treatment plans accordingly. Tongue features such as color, shape, and coating can be utilized to determine if a patient has an underlying health condition. Traditional Chinese tongue diagnosis <xref ref-type="bibr" rid="B15">Solos and Liang (2018)</xref> typically involves observations in the following aspects: 1. Tongue color: Different tongue colors may indicate various health issues. For example, a pale red tongue is often associated with good health, while a deep red tongue may suggest insufficiency of vital energy and blood; 2. Tongue shape: The shape of the tongue can also provide information about the patient&#x2019;s health. For instance, an excessively large tongue, known as a fat and enlarged tongue, often accompanied by tooth imprints, may indicate the insufficiency of both the spleen and the kidney; 3. Tongue coating: The tongue coating, a thin layer of film on the tongue surface, is closely related to the intensity of dampness heat syndrome in TCM theory. Medical studies have shown a correlation between greasy tongue coating and various diseases, such as gastrointestinal disorders, and more recently, the novel coronavirus disease (COVID-19) <xref ref-type="bibr" rid="B13">Pang et al. (2020)</xref>.</p>
<p>Traditional Chinese tongue diagnosis heavily relies on the subjective judgment and clinical experience of TCM practitioners, resulting in outcomes that lack objective indicators <xref ref-type="bibr" rid="B10">Miao et al. (2023)</xref>. The adoption of computer-aided tongue feature recognition models allows for an objective and quantitative diagnosis of tongue conditions, establishing a quantifiable relationship between tongue features and diseases <xref ref-type="bibr" rid="B27">Zhang et al. (2006)</xref>. With significant advancements in computer vision (CV), research on automatic tongue diagnosis systems based on image processing and feature recognition has become more prevalent. For instance, <xref ref-type="bibr" rid="B28">Zhang et al. (2015)</xref> extracted 20 color features and 20 texture features from tongue diagnosis images, including energy, entropy, contrast, and correlation, primarily describing tongue color and coating thickness. <xref ref-type="bibr" rid="B14">Qi et al. (2016)</xref> classified four different tongue colors, employing the ICC profile method for color correction to enhance image consistency. Subsequently, support vector machine (SVM) and random forest (RF) were employed for classification. <xref ref-type="bibr" rid="B12">Pang et al. (2004)</xref> introduced a computerized tongue diagnosis method based on a Bayesian network classifier, focusing on quantitative analysis of tongue color and texture features for diagnostic purposes. <xref ref-type="bibr" rid="B16">Song (2020)</xref> proposed a cascade classifier based on Local Binary Pattern (LBP) features to address the issue of irrelevant information interference, such as lips and cheeks in traditional Chinese tongue diagnosis images. This method utilized LBP features to describe tongue texture and employed the AdaBoost algorithm to construct the cascade classifier. <xref ref-type="bibr" rid="B23">Yamamoto et al. (2011)</xref> utilized a hyperspectral imaging system to acquire tongue images, identifying the most clinically relevant component vectors through Principal Component Analysis (PCA), offering an alternative approach for tongue diagnosis. Additionally, <xref ref-type="bibr" rid="B4">Gao et al. (2007)</xref> employed image processing algorithms to extract quantitative features of the tongue, including color and texture features, and SVM was employed for tongue classification.</p>
<p>However, the complexity of multiple features and variations in tongue image acquisition conditions, such as environmental factors and angles, often render traditional CV algorithms ineffective <xref ref-type="bibr" rid="B22">Xie et al. (2021)</xref>; <xref ref-type="bibr" rid="B8">Li D. et al. (2022)</xref>. With the rapid advancement of deep learning (DL), research on automatic tongue diagnosis programs based on DL models has gained prominence. DL methods typically exhibit stronger generalization and higher feature recognition accuracy compared to traditional computer vision algorithms, circumventing the manual feature extraction drawbacks associated with traditional machine learning methods. Most DL automatic tongue diagnosis systems encompass DL models for both tongue segmentation and tongue feature recognition. Segmentation commonly utilizes models based on U-Net <xref ref-type="bibr" rid="B6">Huang et al. (2020)</xref>, while tongue feature recognition employs pre-trained models such as ResNet or VGG <xref ref-type="bibr" rid="B17">Tammina (2019)</xref>. For instance, <xref ref-type="bibr" rid="B25">Yan J. et al. (2022)</xref> aimed to distinguish different tongue textures, such as the toughness or softness of the tongue body, through the analysis of tongue image textures. They employed the DeepLab v3&#x2b; deep learning semantic segmentation model to segment the tongue image, separating the tongue from the background. Subsequently, a ResNet101-based tongue image texture classification model was constructed. Experimental results demonstrated that using ResNet101 achieved better classification performance compared to traditional tongue image texture classification methods. In another study, <xref ref-type="bibr" rid="B24">Yan B. et al. (2022)</xref> proposed a convolutional neural network based on semantic modeling for tongue segmentation. Different feature extraction networks (AlexNet, VGG16, ResNet18, and DenseNet101) were compared for their effectiveness in extracting tongue color features. Combining U-Net, Inception, and dilated convolutions, <xref ref-type="bibr" rid="B20">Wei et al. (2022)</xref> introduced a new tongue image segmentation method called IAUNet. They designed a network named TCCNet for tongue color classification, incorporating technologies such as ResNet, Inception, and Triplet-Loss. Experimental results showed that TCCNet achieved favorable results in tongue color classification, achieving higher F1-Score and mAP compared to other baselines. Lastly, <xref ref-type="bibr" rid="B9">Li J. et al. (2022)</xref> employed UENET for tongue segmentation, using ResNet34 as the backbone network to extract features and perform classification from tongue photos, with overall accuracy surpassing 86%. <xref ref-type="bibr" rid="B19">Wang et al. (2022)</xref> develop a GreasyCoatNet model based on ResNet, which can recognize and classify different degrees of tongue greasy coating.</p>
<p>These above researches on tongue feature classification are mostly built based on VGG or ResNet <xref ref-type="bibr" rid="B29">Zhuang et al. (2022)</xref>; <xref ref-type="bibr" rid="B7">Huang et al. (2023)</xref>; <xref ref-type="bibr" rid="B9">Li J. et al. (2022)</xref>; <xref ref-type="bibr" rid="B18">Wang et al. (2020)</xref>. However, due to its large size, VGG or ResNet demands substantial computational resources and memory. Additionally, training directly with VGG or ResNet may lead to overfitting, especially when tongue images are challenging to collect and training data is limited. Therefore, our proposed TGANet (Togue Feature Attention Network) utilizes the pretrained VGG16&#x2019;s initial five convolutional blocks as the backbone. Furthermore, we integrate an attention mechanism <xref ref-type="bibr" rid="B3">Fukui et al. (2019)</xref> into the backbone, aiming to mimic human cognitive attention. The primary objective is to focus model weights on crucial parts of the image. For example, in tongue coating classification, the coating is usually concentrated at the root of the tongue. If the model can prioritize local features related to coating, similar to human attention, it enhances efficiency and accuracy. Moreover, the allocation of attention weights during the learning process aids in interpreting causal relationships in the model&#x2019;s judgments. Our proposed architecture TGANet is primarily based on the foundation of <xref ref-type="bibr" rid="B26">Yan et al. (2019)</xref>.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<p>The overall framework for classifying tongue features classification is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>. Initially, the U-Net is employed to segment the input tongue images to obtain the tongue boundary. Subsequently, the masked image derived from the tongue boundary is followed by a data augmentation process. Specifically, the masked image undergoes sequential random rotation, shifting, and adding noise. Following this augmentation, both the masked images and their augmented counterparts are fed into the TGANet to execute the classification of tongue features. In the classification phrase, three kinds of tongue features are classified: tongue color, tongue shape, and tongue coating.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The framework for tongue color recognition encompasses four key stages: Dataset Construction, Segmentation, Data Augmentation, and Classification.</p>
</caption>
<graphic xlink:href="fbioe-12-1392513-g001.tif"/>
</fig>
<sec id="s2-1">
<title>2.1 Dataset Construction</title>
<p>The publicly available BioHit image dataset comprises 300 tongue images with dimensions of 567 &#xd7; 768 pixels. We annotated this original dataset with diagnostic labels. The image annotation process involves three steps. Firstly, domain experts engaged in discussions to establish diagnostic criteria for each category within the three types of tongue features. The detail of the three categories and their class labels is shown in <xref ref-type="table" rid="T1">Table 1</xref>. Subsequently, two well-trained TCM practitioners from the Guangzhou University of Chinese Medicine independently assessed each tongue image to distinguish the class labels for each tongue feature. A third TCM professional with 20 years of expertise joined the deliberations to collectively resolve any disputes and achieve a final consensus. Images with unanimous agreement were then incorporated into the dataset for the development of a deep learning-based tongue feature recognition model.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Tongue feature labels and corresponding descriptions.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Label&#x2216;Tongue feature</th>
<th align="center">Tongue color</th>
<th align="center">Tongue body</th>
<th align="center">Tongue coating</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">0</td>
<td align="center">Pale Red</td>
<td align="center">Swollen</td>
<td align="center">White Greasy</td>
</tr>
<tr>
<td align="center">1</td>
<td align="center">Red</td>
<td align="center">Non-Swollen</td>
<td align="center">Thin White</td>
</tr>
<tr>
<td align="center">2</td>
<td align="center">Dark Red</td>
<td align="center">N/A</td>
<td align="center">Thin Yellow</td>
</tr>
<tr>
<td align="center">3</td>
<td align="center">N/A</td>
<td align="center">N/A</td>
<td align="center">Yellow Greasy</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-2">
<title>2.2 Image segmentation</title>
<p>The aim of tongue image segmentation is to enhance the effectiveness of tongue feature classification by eliminating extraneous information in the image, such as interference from the human jaw or background details, which can disrupt the classification process. To achieve this, we employed the deep convolutional neural network U-Net for tongue segmentation. U-Net is widely utilized in image segmentation, drawing inspiration from semantic segmentation tasks and designed to deliver high-resolution, precise segmentation results.The overall architecture of the U-Net dedicated to segmenting the contour images of the tongue is illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>. U-Net adopts an encoder-decoder structure. The encoder is responsible for sequentially extracting features from the input tongue image through convolution and pooling operations, progressively reducing spatial resolution. The decoder gradually restores spatial resolution through upsampling and deconvolution operations. U-Net incorporates skip connections by linking the output of the last convolutional layer of each encoder block to the corresponding layer in the decoder. This helps retain more detailed information at different resolutions, overcoming potential information loss in deep networks. The final segmentation output is generated in the last layer using a 1 &#xd7; 1 convolutional layer. The training utilizes the cross-entropy loss function to measure the difference between the model&#x2019;s output and the actual segmented image. By leveraging U-Net, we obtain the masked tongue image by acquiring the tongue segmentation contour mask from the input tongue image.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The architecture of U-Net for tongue image segmentation.</p>
</caption>
<graphic xlink:href="fbioe-12-1392513-g002.tif"/>
</fig>
</sec>
<sec id="s2-3">
<title>2.3 Image augmentation</title>
<p>Due to the limited number of samples in medical images and the imbalance in the number of samples for each category, data augmentation is applied to the samples before image classification. This ensures that the quantity of each category in tongue feature classification remains consistent, maintaining an equal number of samples for both the training and validation sets. The commonly employed method to balance categories involves setting the upper limit based on the category with the maximum sample count and augmenting samples from categories with fewer samples.</p>
<p>Various data augmentation techniques are typically utilized, including random translation, random rotation, and the addition of Gaussian noise in different combinations to enhance images. As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, we implemented the augmentation in the order of random translation, followed by random rotation, and then the addition of Gaussian noise. Specifically, random translation involves random shifts in both the <italic>x</italic> and <italic>y</italic>-axes within the range from &#x2212;10 pixels (ps) to 10 ps. Random rotation includes clockwise rotation within the range from &#x2212;15&#xb0; to 15&#xb0;, and Gaussian noise is added with a mean of 0 and a variance of 0.1.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The input tongue images undergo augmentation through the following actions: random shift, random rotation, and the addition of Gaussian noise.</p>
</caption>
<graphic xlink:href="fbioe-12-1392513-g003.tif"/>
</fig>
</sec>
<sec id="s2-4">
<title>2.4 Tongue feature classification</title>
<p>The overall architecture of the TGANet model for tongue feature classification is illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>. We employ VGG16 as the model&#x2019;s backbone, removing all fully connected layers. Input images sequentially pass through convolutional blocks <italic>B</italic>1 to <italic>B</italic>5, extracting global features from the input images. Intermediate features (denoted as <italic>F</italic>) obtained from pooling layers in <italic>B</italic>2 and <italic>B</italic>4 are used to learn attention maps, while the output of the pooling layer after <italic>B</italic>5 (denoted as <italic>G</italic>) represents global features extracted by all convolutional blocks in the network. Intermediate feature <italic>F</italic> and global feature <italic>G</italic> are jointly input into the Attention Module to obtain attention feature <italic>F</italic>:<disp-formula id="e1">
<mml:math id="m1">
<mml:msup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(1)</label>
</disp-formula>Here, &#x201c;Attention&#x201d; represents the operation within the attention module. Specifically, to match the sizes of intermediate and global features, <italic>F</italic> undergoes a convolutional layer to increase its channel count to 256, and bilinear interpolation aligns its feature size with <italic>G</italic>. <italic>G</italic> undergoes a convolutional layer to compress its channel count to 256. The transformed <italic>F</italic> and <italic>G</italic> are then added to obtain <italic>U</italic>:<disp-formula id="e2">
<mml:math id="m2">
<mml:mi>U</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>U</mml:mi>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(2)</label>
</disp-formula>The &#x2a; symbol denotes convolutional operation, and <italic>UP</italic> represents bilinear interpolation. <italic>W</italic>
<sub>
<italic>F</italic>
</sub> and <italic>W</italic>
<sub>
<italic>G</italic>
</sub> are the convolutional weights for <italic>F</italic> and <italic>G</italic>, respectively. Next, <italic>U</italic> undergoes an operation to transform into an attention map <italic>A</italic>:<disp-formula id="e3">
<mml:math id="m3">
<mml:mi>A</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(3)</label>
</disp-formula>Subsequently, pixel-wise multiplication of <italic>F</italic> and <italic>A</italic> yields the Attention Feature:<disp-formula id="e4">
<mml:math id="m4">
<mml:msup>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo>.</mml:mo>
</mml:math>
<label>(4)</label>
</disp-formula>Finally, attention features generated from intermediate features (<italic>B</italic>2 and <italic>B</italic>4) are concatenated with global features. The softmax operation is applied to obtain the final predictions for tongue features. Specifically, predictions are made for three different tongue features: tongue color, tongue shape, and tongue coating. The overall architecture is trained end-to-end.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The architecture of TGANet.The TGANet employs the VGG16 architecture as its backbone, with all fully connected layers removed. Input images undergo sequential processing through convolutional blocks <italic>B</italic>1 to <italic>B</italic>5, capturing global features from the input data. Intermediate features (denoted as <italic>F</italic>) extracted from pooling layers in <italic>B</italic>2 and <italic>B</italic>4 are utilized for attention map learning. Additionally, the output of the pooling layer following <italic>B</italic>5 (denoted as <italic>G</italic>) represents the global features aggregated by all convolutional blocks within the network.</p>
</caption>
<graphic xlink:href="fbioe-12-1392513-g004.tif"/>
</fig>
</sec>
<sec id="s2-5">
<title>2.5 Model evaluation</title>
<p>The model training was conducted on a Windows 11 system equipped with an NVIDIA 4090 GPU, utilizing Python and PyTorch. Initial model parameters were initialized with weights pre-trained on the ImageNet dataset. This transfer learning strategy endowed the model with robust prior knowledge, contributing to superior performance. Subsequently, fine-tuning of the model parameters was performed using the tongue dataset. The parameters of the Attention module were initialized using the Kaiming initialization method, and model optimization employed the Adam optimizer with a learning rate of 0.0001. Three distinct tongue feature classifications shared the same model structure, with the only difference lying in the output of the final fully connected classification layer, which was adjusted according to the different categories. During the training process, model parameter updates were achieved by minimizing the cross-entropy loss function. All models underwent 20 training epochs with a batch size of 20, and model parameters were fixed based on the best performance observed on the validation dataset. Training and testing the model on the tongue color, shape, and coating classification respectively with the 5-fold cross-validation.</p>
</sec>
<sec id="s2-6">
<title>2.6 Metric</title>
<p>Accuracy is the proportion of correctly classified samples by the model on the entire dataset. In model evaluation, accuracy is a crucial metric for assessing the overall performance of the model. The calculation of accuracy <italic>Acc</italic> is the ratio of the number of samples correctly predicted by the model to the total number of samples:<disp-formula id="e5">
<mml:math id="m5">
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:math>
<label>(5)</label>
</disp-formula>where <italic>N</italic>
<sub>
<italic>c</italic>
</sub> is the number of correctly predicted samples, and <italic>N</italic>
<sub>
<italic>t</italic>
</sub> is the total number of samples.</p>
<p>Precision <xref ref-type="bibr" rid="B1">Ashley (2016)</xref> refers to the proportion of actual positive samples among all the samples predicted as positive by the model. In some applications, high precision may be a key objective as it indicates the accuracy of the model in positive class predictions. Precision <italic>P</italic> is calculated as:<disp-formula id="e6">
<mml:math id="m6">
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:math>
<label>(6)</label>
</disp-formula>where <italic>TP</italic> represents true positives, indicating the number of samples correctly predicted as positive by the model, and <italic>FP</italic> represents false positives, indicating the number of instances where the model incorrectly predicted negative class samples as positive.</p>
<p>F1 Score <xref ref-type="bibr" rid="B5">Goutte and Gaussier (2005)</xref> is the harmonic mean of precision and recall, used to comprehensively consider the model&#x2019;s accuracy and recall performance. In some situations, the F1 Score is used as a balance between precision and recall. The calculation of F1 Score <italic>F</italic>1 is given by:<disp-formula id="e7">
<mml:math id="m7">
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:math>
<label>(7)</label>
</disp-formula>where <italic>R</italic> is recall, also known as sensitivity or true positive rate, is a metric that measures the ability of a model to capture all positive instances in the dataset. It is defined as the ratio of <italic>TP</italic> to the sum of <italic>TP</italic> and False Negatives (positive samples incorrectly predicted as negative). The formula for <italic>R</italic> is given by:<disp-formula id="e8">
<mml:math id="m8">
<mml:mi>R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>AUC <xref ref-type="bibr" rid="B21">Wu and Flach (2005)</xref> is the area under the ROC curve, where the ROC curve illustrates the trade-off between true positive and false positive rates at different thresholds. A higher AUC value, closer to 1, indicates better model performance. AUC is commonly used for performance evaluation in binary classification problems, especially when dealing with imbalanced datasets. The specific calculation of AUC is not enumerated here but is typically obtained by integrating the ROC curve.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Baseline models</title>
<p>Throughout the experiments, we compare the TGANet with the following models.<list list-type="simple">
<list-item>
<p>1) VGG16: VGG16 <xref ref-type="bibr" rid="B17">Tammina (2019)</xref> is a deep convolutional neural network architecture designed for image classification tasks. The &#x201c;16&#x2033; in VGG16 refers to the network&#x2019;s depth. VGG16 follows a simple and uniform architecture with small 3 &#xd7; 3 convolutional filters, which helps maintain a consistent receptive field. It also employs max-pooling layers for spatial down-sampling. The pre-trained VGG16 weights on large datasets ImageNet to initialize their models before fine-tuning for our tongue feature classification tasks.</p>
</list-item>
<list-item>
<p>2) ResNet18 <xref ref-type="bibr" rid="B11">Odusami et al. (2021)</xref>: short for Residual Network with 18 layers, is a convolutional neural network architecture introduced by Kaiming He et al. It is part of the ResNet family, known for its deep structure and the incorporation of residual learning blocks. The architecture includes a stack of residual blocks, where each block consists of two convolutional layers with batch normalization and rectified linear unit (ReLU) activation functions. The key innovation in ResNet architectures is the use of skip connections or shortcuts that skip one or more layers, allowing the gradient to flow more easily during backpropagation. This facilitates the training of very deep networks and helps alleviate the vanishing gradient problem. ResNet18 architecture serves as a baseline model and is widely used in tongue feature classification tasks due to its effectiveness and efficiency.</p>
</list-item>
<list-item>
<p>3) TSC-WNet <xref ref-type="bibr" rid="B7">Huang et al. (2023)</xref>: TSC-WNet is a comprehensive neural network architecture designed for the classification of tongue size and shape. TSC-WNet consists of two subnetworks: TSC-UNet and TSC-Net. TSC-Net serves as the classification backbone, while TSC-UNet is responsible for tongue segmentation. TSC-Net employs a simple and efficient architecture with four convolutional layers. By combining both classification and segmentation features, TSC-WNet shows the best validation accuracy and steady performance during training. TSC-WNet is a well-designed network architecture that integrates classification and segmentation tasks, showcasing improved accuracy and robust performance in the challenging domain of tongue analysis.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s3-2">
<title>3.2 Tongue feature classification model performance</title>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> presents a comprehensive performance comparison of various tongue classification models, including ResNet18, TS-WCNet, and our proposed TGANet. The result is the mean and standard deviation of the five folds by using 5-fold cross-validation. The models were evaluated based on different tongue features: Tongue Color, Tongue Shape, and Tongue Coating by using 5-fold cross-validation. For the Tongue Color feature, TGANet outperformed both VGG16, ResNet18, and TSC-WNet with a remarkable accuracy of 91.88%, precision of 90.53%, F1 score of 89.87%, and AUC of 96.45%. These results highlight the superior performance of TGANet in capturing color-related information for tongue classification. Similarly, when focusing on the Tongue Shape feature, TGANet demonstrated a significant improvement in accuracy (92.38%), precision (94.93%), and F1 score (94.05%) compared to VGG16, ResNet18, and TS-WCNet. The robustness of TGANet in extracting shape-related features contributes to its outstanding performance. In the case of Tongue Coating classification, TGANet exhibited outstanding results with an accuracy of 94.77%, precision of 95.59%, and F1 score of 95.02%. This emphasizes the efficacy of TGANet in recognizing and classifying diverse tongue coating patterns. Additionally, the uncertainty in the model&#x2019;s performance is captured through the standard deviation, providing insights into the stability of the results across multiple evaluations. The consistent outperformance of TGANet across different tongue features underscores its robustness and effectiveness in tongue classification tasks.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Comparison of the metrics between our proposed TGANet and the baselines on the three tongue feature classifications (Mean &#xb1; SEM). The result is the mean and standard deviation of the five folds by using 5-fold cross-validation. The best performance is marked in bold.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Tongue feature</th>
<th align="center">Accuracy (%)</th>
<th align="center">Precision (%)</th>
<th align="center">F1 score (%)</th>
<th align="center">AUC (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">VGG16</td>
<td rowspan="4" align="center">Tongue Color</td>
<td align="center">75.13 &#xb1; 4.64</td>
<td align="center">75.43 &#xb1; 5.03</td>
<td align="center">69.16 &#xb1; 5.96</td>
<td align="center">84.31 &#xb1; 3.02</td>
</tr>
<tr>
<td align="center">ResNet18</td>
<td align="center">82.37 &#xb1; 6.42</td>
<td align="center">82.87 &#xb1; 5.40</td>
<td align="center">80.86 &#xb1; 5.78</td>
<td align="center">93.49 &#xb1; 3.10</td>
</tr>
<tr>
<td align="center">TSC-WNet</td>
<td align="center">83.08 &#xb1; 4.68</td>
<td align="center">82.76 &#xb1; 4.92</td>
<td align="center">79.72 &#xb1; 6.45</td>
<td align="center">92.76 &#xb1; 3.59</td>
</tr>
<tr>
<td align="center">TGANet (our)</td>
<td align="center">
<bold>91.88</bold> &#xb1; <bold>2.65</bold>
</td>
<td align="center">
<bold>90.53</bold> &#xb1; <bold>3.16</bold>
</td>
<td align="center">
<bold>89.87</bold> &#xb1; <bold>3.17</bold>
</td>
<td align="center">
<bold>96.45</bold> &#xb1; <bold>1.94</bold>
</td>
</tr>
<tr>
<td align="center">VGG16</td>
<td rowspan="4" align="center">Tongue Shape</td>
<td align="center">91.93 &#xb1; 1.70</td>
<td align="center">94.15 &#xb1; 2.87</td>
<td align="center">93.55 &#xb1; 3.16</td>
<td align="center">96.31 &#xb1; 3.04</td>
</tr>
<tr>
<td align="center">ResNet18</td>
<td align="center">91.43 &#xb1; 2.65</td>
<td align="center">91.15 &#xb1; 2.69</td>
<td align="center">90.25 &#xb1; 2.89</td>
<td align="center">
<bold>97.89</bold> &#xb1; <bold>1.47</bold>
</td>
</tr>
<tr>
<td align="center">TSC-WNet</td>
<td align="center">89.83 &#xb1; 2.00</td>
<td align="center">90.88 &#xb1; 1.84</td>
<td align="center">88.83 &#xb1; 2.85</td>
<td align="center">94.74 &#xb1; 1.80</td>
</tr>
<tr>
<td align="center">TGANet (our)</td>
<td align="center">
<bold>92.38</bold> &#xb1; <bold>1.43</bold>
</td>
<td align="center">
<bold>94.93</bold> &#xb1; <bold>1.63</bold>
</td>
<td align="center">
<bold>94.05</bold> &#xb1; <bold>2.13</bold>
</td>
<td align="center">97.55 &#xb1; 1.57</td>
</tr>
<tr>
<td align="center">VGG16</td>
<td rowspan="4" align="center">Tongue Coating</td>
<td align="center">91.69 &#xb1; 1.41</td>
<td align="center">94.16 &#xb1; 2.40</td>
<td align="center">93.50 &#xb1; 2.61</td>
<td align="center">98.46 &#xb1; 0.70</td>
</tr>
<tr>
<td align="center">ResNet18</td>
<td align="center">90.16 &#xb1; 5.17</td>
<td align="center">93.34 &#xb1; 3.67</td>
<td align="center">91.92 &#xb1; 5.60</td>
<td align="center">
<bold>98.80</bold> &#xb1; <bold>0.62</bold>
</td>
</tr>
<tr>
<td align="center">TSC-WNet</td>
<td align="center">84.62 &#xb1; 2.69</td>
<td align="center">87.32 &#xb1; 3.47</td>
<td align="center">85.54 &#xb1; 3.80</td>
<td align="center">95.87 &#xb1; 1.63</td>
</tr>
<tr>
<td align="center">TGANet (our)</td>
<td align="center">
<bold>94.77</bold> &#xb1; <bold>1.02</bold>
</td>
<td align="center">
<bold>95.59</bold> &#xb1; <bold>1.52</bold>
</td>
<td align="center">
<bold>95.02</bold> &#xb1; <bold>1.72</bold>
</td>
<td align="center">98.77 &#xb1; 1.21</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-3">
<title>3.3 Attention visualization</title>
<p>Attention Visualization is designed to add a visualization of attention weights (attention map) to input images. Initially, the input image is transformed from a PyTorch tensor to a NumPy array, with channel dimensions adjusted to the correct order. The attention map&#x2019;s size is adjusted based on an upsampling factor using bilinear interpolation. The grayscale attention map is then converted to a heatmap using the JET color map from OpenCV. Finally, the image and normalized attention map are blended in a certain proportion, creating an overlay of attention visualization on the image. This process visualizes the depth of focus of a deep learning model on the input. This is particularly helpful in understanding the decision-making process of a deep learning model in tongue feature classification, emphasizing regions considered crucial for tongue segmentation tasks.</p>
<p>As depicted in <xref ref-type="fig" rid="F5">Figure 5</xref>, the attention visualization images for tongue color classification show that the model primarily utilizes features from the tip of the tongue in its decision-making process, which is reasonable given that the color of the tongue tip is typically more distinct. As shown in <xref ref-type="fig" rid="F6">Figure 6</xref>, the attention visualization images for tongue coating classification reveal that the model&#x2019;s decision-making relies heavily on features from the root of the tongue, which is sensible as tongue coating is mainly concentrated in the root area. <xref ref-type="fig" rid="F7">Figure 7</xref> illustrates the attention visualization images for tongue shape classification, demonstrating that the model&#x2019;s decision-making focuses on the contour features of the tongue. This aligns with the common practice among practitioners who assess the thickness and appearance of the tongue&#x2019;s outline to determine its texture.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Visualization of TGANet attention weights for tongue color classification. B2 Attention Maps are the attention weights learning from the intermediate features of <italic>B</italic>2, and B4 Attention Maps are the attention weights learning from the intermediate features of <italic>B</italic>4.</p>
</caption>
<graphic xlink:href="fbioe-12-1392513-g005.tif"/>
</fig>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Visualization of TGANet attention weights for tongue coating classification.</p>
</caption>
<graphic xlink:href="fbioe-12-1392513-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Visualization of TGANet attention weights of TGANet for tongue shape classification.</p>
</caption>
<graphic xlink:href="fbioe-12-1392513-g007.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>TCM practitioners can moniter the patient rehabilitation process by carefully observing and analyzing the patient&#x2019;s tongue feature. For instance, a deep red tongue may suggest a deficiency in vital energy and blood, a fat and enlarged tongue may indicate the insufficiency of both the spleen and the kidney, and a thin layer of film on the tongue surface is closely linked to the intensity of dampness-heat syndrome in TCM theory. Here, we propose a framework designed for the classification of three distinct tongue features. Initially, our expert physicians labeled a publicly available dataset, BioHit, based on these three different tongue features. Subsequently, we preprocessed and augmented the images using image segmentation and augmentation techniques. Then, employing the TGANet architecture with an attention mechanism, we classified the three different tongue features. Our TGANet model outperforms baseline models, achieving the highest accuracy, precision, F1 score, and AUC metrics.</p>
<p>Additionally, the TGANet, based on the VGG16 architecture with attention modules, exhibits superior performance. Compared to the VGG16 without attention modules, the attention modules in our TGANet were further visualized. It was observed that for different tongue feature classifications, the neural network&#x2019;s attention weights varied. For tongue color classification, attention weights were concentrated on the tongue tip; for tongue shape classification, attention weights were focused on the tongue contour; for tongue coating classification, attention weights were centered around the tongue base. This alignment with the expertise of physicians emphasizes the effectiveness of the features learned by our model. Furthermore, the visualization of attention modules provides interpretability for deep learning-based tongue diagnostic models.</p>
<p>In practical applications, establishing a universal model applicable to various tongue feature classifications is highly meaningful in tongue diagnosis and rehabilitation. This contributes to mitigating the overfitting problem. As a result, our TGANet demonstrates outstanding performance in different tongue feature classifications compared to baselines, ultimately leading to more precise diagnoses and better patient rehabilitation in TCM.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In conclusion, our study introduces TGANet, a novel DL model designed for the classification of crucial tongue features in TCM. Leveraging the initial five convolutional blocks of pre-trained VGG16 as the backbone and integrating an Attention mechanism, TGANet outperforms baseline models in accuracy, precision, F1 score, and AUC metrics for distinguishing tongue color, coating, and shape. The integration of an attention mechanism provides interpretability by emphasizing model weights on significant regions of the tongue image. TGANet exhibits robust performance, and the visualization of attention weight further reveals the model&#x2019;s focus on specific tongue regions for decision-making, aligning with clinical practices. This study contributes to advancing automatic tongue diagnosis systems, providing a foundation for objective and quantitative assessment of tongue conditions in TCM.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>ZY: Writing&#x2013;original draft, Writing&#x2013;review and editing, Conceptualization, Investigation, Supervision. XL: Data curation, Formal Analysis, Funding acquisition, Project administration, Resources, Supervision, Writing&#x2013;review and editing. AS: Data curation, Funding acquisition, Methodology, Project administration, Resources, Supervision, Writing&#x2013;review and editing. SF: Conceptualization, Data curation, Formal Analysis, Methodology, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>Author AS was employed by Guangdong Zhengyuanchun Traditional Chinese Medicine Clinic Co., Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ashley</surname>
<given-names>E. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Towards precision medicine</article-title>. <source>Nat. Rev. Genet.</source> <volume>17</volume>, <fpage>507</fpage>&#x2013;<lpage>522</lpage>. <pub-id pub-id-type="doi">10.1038/nrg.2016.86</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Du</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Multifunctional coatings of nickel-titanium implant toward promote osseointegration after operation of bone tumor and clinical application: a review</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>12</volume>, <fpage>1325707</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2024.1325707</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Fukui</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hirakawa</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yamashita</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Fujiyoshi</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Attention branch network: learning of attention mechanism for visual explanation</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>10705</fpage>&#x2013;<lpage>10714</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Po</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2007</year>). &#x201c;<article-title>A novel computerized method based on support vector machine for tongue diagnosis</article-title>,&#x201d; in <conf-name>2007 third international IEEE conference on signal-image technologies and internet-based system</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>849</fpage>&#x2013;<lpage>854</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Goutte</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gaussier</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>A probabilistic interpretation of precision, recall and f-score, with implication for evaluation</article-title>,&#x201d; in <conf-name>European conference on information retrieval</conf-name> (<publisher-name>Springer</publisher-name>), <fpage>345</fpage>&#x2013;<lpage>359</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Iwamoto</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Unet 3&#x2b;: a full-scale connected unet for medical image segmentation</article-title>,&#x201d; in <conf-name>ICASSP 2020-2020 IEEE international conference on acoustics, speech and signal processing (ICASSP)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>1055</fpage>&#x2013;<lpage>1059</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Tongue size and shape classification fusing segmentation features for traditional Chinese medicine diagnosis</article-title>. <source>Neural Comput. Appl.</source> <volume>35</volume>, <fpage>7581</fpage>&#x2013;<lpage>7594</lpage>. <pub-id pub-id-type="doi">10.1007/s00521-022-08054-y</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yin</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Deep learning and machine intelligence: new computational modeling techniques for discovery of the combination rules and pharmacodynamic characteristics of traditional Chinese medicine</article-title>. <source>Eur. J. Pharmacol.</source> <volume>933</volume>, <fpage>175260</fpage>. <pub-id pub-id-type="doi">10.1016/j.ejphar.2022.175260</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022b</year>). <article-title>Automatic classification framework of tongue feature based on convolutional neural networks</article-title>. <source>Micromachines</source> <volume>13</volume>, <fpage>501</fpage>. <pub-id pub-id-type="doi">10.3390/mi13040501</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Image recognition of traditional Chinese medicine based on deep learning</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>11</volume>, <fpage>1199803</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2023.1199803</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Odusami</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Maskeliunas</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Dama&#x161;evi&#x10d;ius</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Krilavi&#x10d;ius</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Analysis of features of alzheimer&#x2019;s disease: detection of early stage from functional brain changes in magnetic resonance images using a finetuned resnet18 network</article-title>. <source>Diagnostics</source> <volume>11</volume>, <fpage>1071</fpage>. <pub-id pub-id-type="doi">10.3390/diagnostics11061071</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Computerized tongue diagnosis based on bayesian networks</article-title>. <source>IEEE Trans. Biomed. Eng.</source> <volume>51</volume>, <fpage>1803</fpage>&#x2013;<lpage>1810</lpage>. <pub-id pub-id-type="doi">10.1109/tbme.2004.831534</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Tongue features of patients with coronavirus disease 2019: a retrospective cross-sectional study</article-title>. <source>Integr. Med. Res.</source> <volume>9</volume>, <fpage>100493</fpage>. <pub-id pub-id-type="doi">10.1016/j.imr.2020.100493</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Tu</surname>
<given-names>L.-p.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.-b.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X.-j.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.-t.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.-f.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>The classification of tongue colors with standardized acquisition and icc profile correction in traditional Chinese medicine</article-title>. <source>BioMed Res. Int.</source> <volume>2016</volume>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1155/2016/3510807</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Solos</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A historical evaluation of Chinese tongue diagnosis in the treatment of septicemic plague in the pre-antibiotic era, and as a new direction for revolutionary clinical research applications</article-title>. <source>J. Integr. Med.</source> <volume>16</volume>, <fpage>141</fpage>&#x2013;<lpage>146</lpage>. <pub-id pub-id-type="doi">10.1016/j.joim.2018.04.001</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Tongue localization method based on cascade classifier</article-title>. <source>J. Artif. Intell. Pract.</source> <volume>3</volume>, <fpage>13</fpage>&#x2013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.23977/jaip.2020.030104</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tammina</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Transfer learning using vgg-16 with deep convolutional neural network for classifying images</article-title>. <source>Int. J. Sci. Res. Publ. (IJSRP)</source> <volume>9</volume>, <fpage>94200</fpage>&#x2013;<lpage>p10150</lpage>. <pub-id pub-id-type="doi">10.29322/ijsrp.9.10.2019.p9420</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Artificial intelligence in tongue diagnosis: using deep convolutional neural network for recognizing unhealthy tongue with tooth-mark</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>18</volume>, <fpage>973</fpage>&#x2013;<lpage>980</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2020.04.002</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Huo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Constructing tongue coating recognition model using deep transfer learning to assist syndrome diagnosis and its potential in noninvasive ethnopharmacological evaluation</article-title>. <source>J. Ethnopharmacol.</source> <volume>285</volume>, <fpage>114905</fpage>. <pub-id pub-id-type="doi">10.1016/j.jep.2021.114905</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jinming</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xingjin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Hui</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Tongue image segmentation and tongue color classification based on deep learning</article-title>. <source>Digit. Chin. Med.</source> <volume>5</volume>, <fpage>253</fpage>&#x2013;<lpage>263</lpage>. <pub-id pub-id-type="doi">10.1016/j.dcmed.2022.10.002</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Flach</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>A scored auc metric for classifier evaluation and selection</article-title>,&#x201d; in <source>Second workshop on ROC analysis in ML</source> (<publisher-loc>bonn, Germany</publisher-loc>).</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jing</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Digital tongue image analyses for health assessment</article-title>. <source>Med. Rev.</source> <volume>1</volume>, <fpage>172</fpage>&#x2013;<lpage>198</lpage>. <pub-id pub-id-type="doi">10.1515/mr-2021-0018</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yamamoto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tsumura</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Nakaguchi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Namiki</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kasahara</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ogawa-Ochiai</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Principal component vector rotation of the tongue color spectrum to predict &#x201c;mibyou&#x201d;(disease-oriented state)</article-title>. <source>Int. J. Comput. assisted radiology Surg.</source> <volume>6</volume>, <fpage>209</fpage>&#x2013;<lpage>215</lpage>. <pub-id pub-id-type="doi">10.1007/s11548-010-0506-8</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Tongue segmentation and color classification using deep convolutional neural networks</article-title>. <source>Mathematics</source> <volume>10</volume>, <fpage>4286</fpage>. <pub-id pub-id-type="doi">10.3390/math10224286</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2022b</year>). <article-title>Tongue image texture classification based on image inpainting and convolutional neural network</article-title>. <source>Comput. Math. Methods Med.</source> <volume>2022</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1155/2022/6066640</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kawahara</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hamarneh</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Melanoma recognition via visual attention</article-title>,&#x201d; in <conf-name>Information Processing in Medical Imaging: 26th International Conference, IPMI 2019</conf-name>, <conf-loc>Hong Kong, China</conf-loc>, <conf-date>June 2&#x2013;7, 2019</conf-date> (<publisher-name>Springer</publisher-name>), <fpage>793</fpage>&#x2013;<lpage>804</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>Computer aided tongue diagnosis system</article-title>,&#x201d; in <conf-name>2005 IEEE engineering in medicine and biology 27th annual conference</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>6754</fpage>&#x2013;<lpage>6757</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Preliminary study of tongue image classification based on multi-label learning</article-title>,&#x201d; in <conf-name>Proceedings, Part III 11 Advanced Intelligent Computing Theories and Applications: 11th International Conference, ICIC 2015</conf-name>, <conf-loc>Fuzhou, China</conf-loc>, <conf-date>August 20-23, 2015</conf-date> (<publisher-name>Springer</publisher-name>), <fpage>208</fpage>&#x2013;<lpage>220</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhuang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Gan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Human-computer interaction based health diagnostics using resnet34 for tongue image classification</article-title>. <source>Comput. Methods Programs Biomed.</source> <volume>226</volume>, <fpage>107096</fpage>. <pub-id pub-id-type="doi">10.1016/j.cmpb.2022.107096</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>