<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2023.1212049</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>STNet: shape and texture joint learning through two-stream network for knowledge-guided image recognition</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Wang</surname> <given-names>Xijing</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2136949/overview"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Han</surname> <given-names>Hongcheng</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Xu</surname> <given-names>Mengrui</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Shengpeng</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Dong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1606327/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Du</surname> <given-names>Shaoyi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1476228/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Xu</surname> <given-names>Meifeng</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/441896/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>National Key Laboratory of Human-Machine Hybrid Augmented Intelligence, National Engineering Research Center for Visual Information and Applications, Institute of Artificial Intelligence and Robotics, Xi&#x00027;an Jiaotong University</institution>, <addr-line>Xi&#x00027;an</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>The School of Software Engineering, Xi&#x00027;an Jiaotong University</institution>, <addr-line>Xi&#x00027;an</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>The School of Automation Science and Engineering, Xi&#x00027;an Jiaotong University</institution>, <addr-line>Xi&#x00027;an</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>The Second Affiliated Hospital of Xi&#x00027;an Jiaotong University (Xibei Hospital)</institution>, <addr-line>Xi&#x00027;an</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Zhengwang Wu, University of North Carolina at Chapel Hill, United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Ling Ma, Nankai University, China; Hailong Huang, Hong Kong Polytechnic University, Hong Kong SAR, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Shaoyi Du <email>dushaoyi&#x00040;gmail.com</email></corresp>
<corresp id="c002">Meifeng Xu <email>xumf96&#x00040;163.com</email></corresp>
<fn fn-type="equal" id="fn002"><p>&#x02020;These authors have contributed equally to this work</p></fn></author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>06</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>17</volume>
<elocation-id>1212049</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>04</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>05</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2023 Wang, Han, Xu, Li, Zhang, Du and Xu.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Wang, Han, Xu, Li, Zhang, Du and Xu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>The human brain processes shape and texture information separately through different neurons in the visual system. In intelligent computer-aided imaging diagnosis, pre-trained feature extractors are commonly used in various medical image recognition methods, common pre-training datasets such as ImageNet tend to improve the texture representation of the model but make it ignore many shape features. Weak shape feature representation is disadvantageous for some tasks that focus on shape features in medical image analysis.</p>
</sec>
<sec>
<title>Methods</title>
<p>Inspired by the function of neurons in the human brain, in this paper, we proposed a shape-and-texture-biased two-stream network to enhance the shape feature representation in knowledge-guided medical image analysis. First, the two-stream network shape-biased stream and a texture-biased stream are constructed through classification and segmentation multi-task joint learning. Second, we propose pyramid-grouped convolution to enhance the texture feature representation and introduce deformable convolution to enhance the shape feature extraction. Third, we used a channel-attention-based feature selection module in shape and texture feature fusion to focus on the key features and eliminate information redundancy caused by feature fusion. Finally, aiming at the problem of model optimization difficulty caused by the imbalance in the number of benign and malignant samples in medical images, an asymmetric loss function was introduced to improve the robustness of the model.</p>
</sec>
<sec>
<title>Results and conclusion</title>
<p>We applied our method to the melanoma recognition task on ISIC-2019 and XJTU-MM datasets, which focus on both the texture and shape of the lesions. The experimental results on dermoscopic image recognition and pathological image recognition datasets show the proposed method outperforms the compared algorithms and prove the effectiveness of our method.</p>
</sec></abstract>
<kwd-group>
<kwd>computer-aided diagnosis</kwd>
<kwd>image recognition</kwd>
<kwd>feature fusion</kwd>
<kwd>joint learning</kwd>
<kwd>two-stream network</kwd>
<kwd>brain-like information processing</kwd>
</kwd-group>
<contract-num rid="cn001">62173269</contract-num>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content></contract-sponsor>
<counts>
<fig-count count="8"/>
<table-count count="5"/>
<equation-count count="13"/>
<ref-count count="57"/>
<page-count count="13"/>
<word-count count="8700"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Neuroprosthetics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1. Introduction</title>
<p>Computer-aided diagnosis (CAD) has been a research hotspot for the past few decades. CAD automatically analyzes the patient data through machine learning methods to make an assessment of the patient&#x00027;s condition (Yanase and Triantaphyllou, <xref ref-type="bibr" rid="B46">2019</xref>; Chan et al., <xref ref-type="bibr" rid="B6">2020</xref>). Medical image analysis is one of the most important fields in CAD technologies, it helps read imaging data to improve the diagnosis efficiency. An intelligent medical image analysis model can share the workload of radiologists and pathologists, and enables areas with underdeveloped medical resources to achieve high-level imaging analysis at low cost (Shen et al., <xref ref-type="bibr" rid="B38">2017</xref>; Kurc et al., <xref ref-type="bibr" rid="B23">2020</xref>).</p>
<p>In the past decade, medical image analysis methods have grown by leaps and bounds due to the development of deep learning and computer vision algorithms. Powerful feature representation ability enables deep neural networks to learn complex hidden features from a large amount of training data, which overcomes the difficulty of manual feature design in traditional medical image analysis methods. However, there are still challenges to be addressed in current deep learning-based algorithms for medical image analysis, with weak shape representation being one of the most critical issues. On the one hand, in the commonly used convolutional neural network (CNN), the limited receptive field of kernels tends to fit local features during kernel parameter learning. Although the range of the receptive field of deep convolutional kernels on original images gradually increases as layers deepen, deeper layers weaken their connection with original images, which limits networks in modeling shape features at larger scales (Luo et al., <xref ref-type="bibr" rid="B30">2016</xref>; Araujo et al., <xref ref-type="bibr" rid="B4">2019</xref>). On the other hand, pre-trained parameters are frequently employed in medical image recognition techniques to expedite convergence during training and potentially enhance model performance. Given the paucity of annotated data in medical images, large-scale natural image datasets such as ImageNet (Deng et al., <xref ref-type="bibr" rid="B10">2009</xref>; Russakovsky et al., <xref ref-type="bibr" rid="B37">2015</xref>) are commonly utilized as pre-training datasets. However, the research of Geirhos et al. (<xref ref-type="bibr" rid="B13">2018</xref>) indicates that the deep neural network pre-trained on ImageNet is biased to focus on the texture features and has relatively weak shape feature representation ability.</p>
<p>The weak representation of shapes, caused by the limitations of the model and pre-training datasets, significantly impacts the performance of the model on certain shape-dependent medical image tasks. As, <xref ref-type="fig" rid="F1">Figure 1</xref> shows, cascade segmentation and classification model (Chang, <xref ref-type="bibr" rid="B7">2017</xref>) can solve the problem in some scenarios, it uses a segmentation network to obtain the mask of a lesion, and then use the segmented lesion image as the input of the classification network, providing shape information for classification, eliminating the background noise. However, the lack of sufficient training data is a prevalent issue in various medical image analysis tasks, resulting in inadequate precision of the trained segmentation task. Inaccurate segmentation can provide erroneous shape information for classification. In addition, the cascade segmentation and classification model contains two encoders and one decoder, and they are cascaded, the research of He et al. (<xref ref-type="bibr" rid="B17">2017</xref>) indicates that repetitive encoding and decoding operations yield minimal improvements to the quality of extracted features.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Weak feature representation problem of many existing methods for image recognition in computer-aided diagnosis. <bold>(A)</bold> Common image recognition model. <bold>(B)</bold> Cascade segmentation and classification model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-17-1212049-g0001.tif"/>
</fig>
<p>In order to solve the above problems, we proposed a shape-and-texture-biased two-stream network to enhance the shape feature representation in knowledge-guided medical image analysis. The human brain processes shape and texture information separately through different neurons in the visual system, inspired by that, first, the two-stream network shape-biased stream and a texture-biased stream are constructed through classification and segmentation multi-task joint learning. Second, we propose pyramid-grouped convolution (PGC) to enhance the texture feature representation, and introduce deformable convolution (DC) to enhance the shape feature extraction. Third, we used a channel-attention-based feature selection module in shape and texture feature fusion to focus on the key features and eliminate information redundancy caused by feature fusion. Finally, aiming at the problem of model optimization difficulty caused by the imbalance in the number of benign and malignant samples in medical images, an asymmetric loss function was introduced to improve the robustness of the model. We applied our method to the melanoma recognition task on ISIC-2019 (Rotemberg et al., <xref ref-type="bibr" rid="B36">2021</xref>) and XJTU-MM datasets, which focuses on both the texture and shape of the lesions. The experimental results on dermatoscopic image recognition and pathological image recognition show that the proposed method outperforms the compared algorithms and prove the effectiveness of our method.</p>
<p>The main contributions of this work are enumerated as follows:</p>
<list list-type="bullet">
<list-item><p>We propose the shape and texture joint learning two-stream network for knowledge-guided medical image recognition, taking into account the learning of shape features and texture features by the network, addressing the weak shape representation problem of existed methods.</p></list-item>
<list-item><p>We propose pyramid-grouped convolution to enhance the texture feature representation, and introduce deformable convolution to address the limitation of fixed respective fields, enhancing the shape feature extraction.</p></list-item>
<list-item><p>We construct the shape and texture fusion module based on channel attention mechanism to focus on the essential features and eliminate the noise, reducing the information redundancy caused by feature fusion.</p></list-item>
<list-item><p>We introduce the asymmetric loss function for optimization, reducing the impact of commonly existed sample imbalance problem in medical image datasets.</p></list-item>
</list>
</sec>
<sec id="s2">
<title>2. Related work</title>
<sec>
<title>2.1. Knowledge-guided medical image analysis</title>
<p>Most of the key technologies in medical image analysis come from general computer vision algorithms, however, the image characteristics and the data distribution are different between natural images and medical images. Constructing appropriate deep neural network model with the guidance of the prior knowledge from pathology and radiology is important for improving model performance in specific medical analysis tasks.</p>
<p>Fan et al. (<xref ref-type="bibr" rid="B11">2017</xref>) proposed a novel automatic segmentation algorithm using saliency combined with Otsu threshold for dermoscopy images, which extracted prior information on healthy skin to construct the color saliency map and brightness saliency map respectively. Ahn et al. (<xref ref-type="bibr" rid="B1">2017</xref>) proposed a saliency-based lesion segmentation method in dermoscopic images, using the reconstruction errors derived from a sparse representation model coupled with a novel background detection. Yang et al. (<xref ref-type="bibr" rid="B48">2023</xref>) proposed a Multi-scale Fully-shared Fusion Network (MFF-Net) that gathers features of dermoscopic images and clinical images for skin lesion classification. Zhang et al. (<xref ref-type="bibr" rid="B53">2018a</xref>) used deep learning algorithms to help diagnose four common cutaneous diseases based on dermoscopic images and summarized classification/diagnosis scenarios based on domain expert knowledge and semantically represented them in a hierarchical structure to improve the accuracy of the algorithm. Clinical prior knowledge is also widely applied to the analysis of ultrasound images and other medical images. Liu et al. (<xref ref-type="bibr" rid="B25">2019b</xref>) proposed a novel deep-learning-based CAD system, guided by task-specific prior knowledge, for automated nodule detection and classification in ultrasound images. Chen et al. (<xref ref-type="bibr" rid="B8">2021</xref>) proposed a knowledge-guided data augmentation framework for breast lesion classification, which consists of a modal translater and a semantic inverter, achieving cross-modal and semantic data augmentation simultaneously. Shi et al. (<xref ref-type="bibr" rid="B39">2020</xref>) proposed a knowledge-guided synthetic medical image adversarial augmentation method for ultrasonography thyroid nodule classification, extracting domain knowledge from standardized terminology to improve the classification performance. Yang et al. (<xref ref-type="bibr" rid="B47">2021</xref>) proposed a multi-task cascade deep learning model (MCDLM), which integrates radiologists&#x00027; various domain knowledge (DK) and used multimodal ultrasound images for automatic diagnosis of thyroid nodules. Han et al. (<xref ref-type="bibr" rid="B16">2020</xref>) proposed an ensemble learning method for panoramic radiographs recognition based on the characteristics of each stage of tooth growth. Ni et al. (<xref ref-type="bibr" rid="B32">2013</xref>) proposed a novel learning-based automatic method to detect the fetal head for the measurement of head circumference from ultrasound images and used prior knowledge and online imaging parameters to guide the sliding window-based head detection. Pan et al. (<xref ref-type="bibr" rid="B34">2022</xref>) proposed a two-stage network with prior knowledge guidance for medullary thyroid carcinoma recognition in ultrasound images. Meanwhile, extracting and fusing semantic features of solid tissues and calcification for better recognizing the segmented nodules. Zhou et al. (<xref ref-type="bibr" rid="B57">2022</xref>) proposed a rheumatoid arthritis knowledge-guided (RATING) system for scoring rheumatoid arthritis activity from multimodal ultrasound images, leveraging diagnostic paradigm and experience to enhance the robustness. Lu et al. (<xref ref-type="bibr" rid="B28">2023</xref>) proposed a Prior Knowledge-based Relation Transformer Network (PKRT-Net), which employed the clinical prior knowledge to assist OC segmentation. Gao et al. (<xref ref-type="bibr" rid="B12">2021</xref>) proposed a medical-knowledge-guided one-class classification approach that leverages domain-specific knowledge of classification tasks to boost the model&#x00027;s performance and showed superior model performance on three different clinical image classification tasks. Zhang et al. (<xref ref-type="bibr" rid="B50">2023</xref>) proposed coarse-to-fine method for melanoma and nevi recognition according to distribution of inter-class and intra-class differences as summarized by dermatologists.</p>
<p>Prior knowledge provides inspiration for medical image analysis design, in this paper, we innovate a novel method for shape-relied medical image recognition.</p>
</sec>
<sec>
<title>2.2. Shape and texture feature fusion</title>
<p>Aiming at the problem of weak shape representation of existing CNN-based medical image recognition models, we investigate the texture and shape feature fusion algorithms designed for various tasks.</p>
<p>Al-Osaimi et al. (<xref ref-type="bibr" rid="B2">2011</xref>) proposed spatially optimized data/pixel-level fusion of 3-D shape and texture for face recognition. Lu et al. (<xref ref-type="bibr" rid="B29">2017</xref>) proposed a face image retrieval method based on shape and texture feature fusion, which used accurate facial landmark locations as shape features and utilized shape priors to provide discriminative texture features. Kotsia et al. (<xref ref-type="bibr" rid="B22">2008</xref>) proposed a novel method based on the fusion of texture and shape information for facial expression and Facial Action Unit (FAU) recognition from video sequences and used various approaches to perform texture and shape feature fusion, among which were SVMs and Median Radial Basis Functions (MRBFs). Anantharatnasamy et al. (<xref ref-type="bibr" rid="B3">2013</xref>) proposed a content-based image retrieval system based on three major types of visual information including color, texture, shape, and their distances to the origin in a three dimensional space for the retrieval. Sumathi and Kumar (<xref ref-type="bibr" rid="B40">2012</xref>) extracted edge and texture features using Gabor filter and fused them for plant leaf classification. Xiong et al. (<xref ref-type="bibr" rid="B44">2007</xref>) proposed a Statistical Shape and Radio texture fusion model for facial expression sequence synthesis, processing facial shape and texture separately and fusing them together to synthesize the final result. Jo et al. (<xref ref-type="bibr" rid="B21">2014</xref>) proposed a new method for eye state classification to detect diver drowsiness, which extracted and fused features from both eyes. Zhang et al. (<xref ref-type="bibr" rid="B55">2020</xref>) proposed two-stream networks to enhance the extraction of shape and texture respectively for clothing classification and attribute recognition.</p>
<p>These researches use various of methods to enhance the texture and shape feature learning on specific data. For shape-relied medical image recognition tasks, we design the model to realize that with the guidance of the prior knowledge, such as visual characteristics and category distribution.</p>
</sec>
</sec>
<sec sec-type="methods" id="s3">
<title>3. Methodology</title>
<sec>
<title>3.1. Framework</title>
<p>In contrast to the cascade segmentation and classification model, our proposed model employs a two-stream network for joint learning of shape and texture, mitigating the impact of imprecise segmentation on shape information in the former. The overall framework of the proposed method is shown as <xref ref-type="fig" rid="F2">Figure 2</xref>, the input image is fed into the parallel texture-biased stream and shape-biased stream. First, the texture-biased stream consists of a feature encoder, which is pre-trained on texture-biased large-scale dataset, such as ImageNet. To further enhance the texture feature representation ability of the texture feature encoder, we reconstruct the convolutional block using the proposed channel connection pyramid mechanism. Second, the shape-biased stream contains an encoder-decoder based network, the encoder extracts the shape features and the decoder generates the lesion mask, the quality of the extracted shape features is supervised by <italic>L</italic>2 loss function between the predicted mask and the ground truth mask. Third, the texture feature and the shape feature are concatenated and input to the feature fusion module, to address the information redundancy problem in feature fusion, we construct the feature fusion module based on channel attention mechanism to focus on the essential features and eliminate the effects of noise. In addition, to balance the texture-biased learning and shape-biased learning, the gradient scaling layer is added between the shape feature map and the concatenation operation to weight the gradient in the back propagation. Then, the fully connected layer classifier is used to output the classification results. Finally, to overcome the optimization difficulty caused by the problem of imbalanced samples in medical image datasets, we introduce the asymmetric loss to enhance the attention of the model to the categories with smaller numbers of samples.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Framework of the proposed shape and texture joint learning two-stream network. <bold>(A)</bold> Texture-biased stream. <bold>(B)</bold> Shape-biased stream. <bold>(C)</bold> Feature fusion module. <bold>(D)</bold> Classifier. <bold>(E)</bold> Asymmetric loss.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-17-1212049-g0002.tif"/>
</fig>
</sec>
<sec>
<title>3.2. Texture-biased stream</title>
<p>The texture-biased stream is constructed by the texture feature encoder pre-trained on texture-biased dataset ImageNet. To enhance the texture feature representation, we improve the channel connections in convolutional blocks. In the standard convolution operation, each kernel is connected to every channel of the input feature map. However, while the large number of learnable parameters provides a powerful fitting ability for the network, overly dense connections can lead to significant information redundancy and unnecessary computational burden (Huang et al., <xref ref-type="bibr" rid="B20">2017</xref>; Ma et al., <xref ref-type="bibr" rid="B31">2018</xref>; Zhang et al., <xref ref-type="bibr" rid="B54">2018b</xref>). Grouped convolution mechanism (Xie et al., <xref ref-type="bibr" rid="B43">2017</xref>; Zhang H. et al., <xref ref-type="bibr" rid="B51">2022</xref>) provides an efficient way to solve the problem, it divides the input feature map into several groups in the channel dimension, each kernel has connections to the specific group only rather than all channels of the input feature map. With the same number of output feature map channels, channel-wise connections become sparser, thereby enhancing diagonal correlations between channels. Depth-wise convolution (Chollet, <xref ref-type="bibr" rid="B9">2017</xref>) even makes the connections more sparse, which regards each channel of the input feature map as one group to perform grouped convolution. With fewer learnable kernel parameters, depth-wise convolution even shows stronger low-level texture feature representation ability (Guo et al., <xref ref-type="bibr" rid="B14">2019</xref>; Tan and Le, <xref ref-type="bibr" rid="B41">2019</xref>). However, grouped convolution and depth-wise convolution still have problems in balancing the learning of low-level and high-level texture features.</p>
<p>To further improve the feature extraction quality and efficiency, we propose the pyramid-grouped convolution(PGC) mechanism to enhance the feature representation of the texture-biased stream. As <xref ref-type="fig" rid="F3">Figure 3</xref> shows, In each pyramid-convolutional block, the density of channel connections varies layer by layer, transitioning from dense to sparse. This results in a transition of the channel-wise receptive field of each kernel from large to small, leading to sparser feature encoding compared to conventional grouped convolution and more appropriate channel-wise receptive fields than depth-wise convolution. The PGC blocks are embedded in the backbone network to construct feature encoder of texture-biased stream, enhancing the texture feature representation.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Pyramid-grouped convolution. In each pyramid, the density of channel connection changes layer by layer, and from dense to sparse.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-17-1212049-g0003.tif"/>
</fig>
</sec>
<sec>
<title>3.3. Shape-biased stream</title>
<p>Pixel-wise semantic segmentation model is a learning paradigm conducive to modeling shape features (Long et al., <xref ref-type="bibr" rid="B27">2015</xref>; Guo et al., <xref ref-type="bibr" rid="B15">2018</xref>). In the proposed method, the shape-biased stream is constructed using an encoder-decoder based segmentation network, the decoder generates the lesion mask based on the features extracted from the input image. With the supervision of the <italic>L</italic>2 loss between the predicted mask and the ground truth mask, the encoder is encouraged to learn the shape-biased features. Many encoder-decoder based semantic segmentation models add shortcut connections between encoders and decoders to enhance the contributions of low-level features extracted by shallow layers in encoders to mask generation, which are usually called U-shape networks (Ronneberger et al., <xref ref-type="bibr" rid="B35">2015</xref>; Oktay et al., <xref ref-type="bibr" rid="B33">2018</xref>; Zhou et al., <xref ref-type="bibr" rid="B56">2018</xref>; Zhang et al., <xref ref-type="bibr" rid="B52">2021</xref>). But in the shape-biased stream of our method, all we need is to improve the shape feature representation of the feature map extracted by feature encoder, all the information flow is expected to pass through the deepest feature map, so we did not add any shortcut connection between the encoder and the decoder.</p>
<p>In the design of the shape encoder network, we introduce the deformable kernel to address the limitation of the rectangular receptive field of the convolution kernel. Irregular-shaped visual features are common in lesion images, for example, the irregular-shape boundary of the lesion in dermoscopic images (Celebi et al., <xref ref-type="bibr" rid="B5">2019</xref>), the irregular-shaped cells in pathological images (Zhang D. et al., <xref ref-type="bibr" rid="B49">2022</xref>). Rectangular convolutional kernels have limitation in extracting these features, especially in extracting low-level shape features. As <xref ref-type="fig" rid="F4">Figure 4</xref> shows, the discrete feature map is regarded as a continuous two-dimensional distribution, we insert an offset layer to learn a offset to transform the rectangular kernel to an kernel with irregular shape that better match the extracted features. The feature map in the deformable receptive field is resampled through bilinear interpolation according to the parameters of the learned offset. deformable convolution is calculated by</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold-italic"><mml:mi>y</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mstyle mathvariant="bold-italic"><mml:mi>R</mml:mi></mml:mstyle></mml:mrow></mml:munder></mml:mstyle><mml:mstyle mathvariant="bold-italic"><mml:mi>w</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold-italic"><mml:mi>x</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x00394;</mml:mtext><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic><bold>y</bold></italic>(<italic>p</italic>) indicates the feature obtained by the convolution on one sampling point <italic>p</italic> of the feature map. <italic><bold>R</bold></italic> is the receptive field size of the regular kernel. <italic>p</italic><sub><italic>k</italic></sub> donates the difference between the sampling points and <italic><bold>y</bold></italic>(<italic>p</italic>), <italic>k</italic> &#x0003D; 1, 2, 3...<italic>N, N</italic> &#x0003D; |<italic><bold>R</bold></italic>|, &#x00394;<italic>p</italic><sub><italic>k</italic></sub> is the learned offset, and <italic><bold>w</bold></italic> is the kernel parameter. We reconstruct the backbone network of feature encoder using deformable convolution layers, enhancing the representation of irregular-shaped features.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Deformable convolution. <bold>(A)</bold> Deformable kernel. <bold>(B)</bold> Deformable convolutional layer. An offset layer is inserted to learn the offset to transform the rectangular kernel to a kernel with an irregular shape that better match the extracted features. The feature map in the deformable receptive field is resampled through bilinear interpolation according to the parameters of the learned offset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-17-1212049-g0004.tif"/>
</fig>
</sec>
<sec>
<title>3.4. Channel-attention-based texture and shape feature fusion</title>
<p>The feature maps extracted from the texture-biased and shape-biased streams are concatenated to fuse texture and shape features, which expands the scope of the extracted features. However, this also results in a certain degree of information redundancy. Some irrelevant features not only fail to contribute to improving model performance but also increase the risk of overfitting and negatively impact model robustness. To select essential features for lesion recognition and eliminate irrelevant features and noise, we design the texture and shape feature fusion module based on channel attention mechanism.</p>
<p>Each kernel represents a specific hidden feature, having a specific correlation with lesion recognition, feature selection is equivalent to kernel selection, which can also be regarded as the selection of channels of feature map. We introduce the channel attention mechanism to highlight the essential channels and suppress noise through learning the channel weights based on the global representation of each channel. As <xref ref-type="fig" rid="F5">Figure 5</xref> shows, for the <italic>w</italic>&#x000D7;<italic>h</italic>&#x000D7;<italic>c</italic> input feature map <bold>Z</bold>, it is first transformed into a 1 &#x000D7; 1 &#x000D7; <italic>c</italic> feature vector <italic><bold>g</bold></italic> through global pooling, which combines average pooling and max pooling to balance average and peak characterization, calculating by</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">max</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>z</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Channel attention mechanism. The attention weight vector <italic><bold>w</bold></italic><sub><italic>att</italic></sub> is calculated through global pooling and 1 &#x000D7; 1 convolutional layers, then the input feature map Z is weighted to obtain the output feature map Z&#x02032;.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-17-1212049-g0005.tif"/>
</fig>
<p>where <italic>g</italic><sub><italic>k</italic></sub> is the element in feature vector <italic><bold>g</bold></italic>, <italic>z</italic><sub><italic>i,j,k</italic></sub> is the element in <italic>k</italic>-th channel of feature map <bold>Z</bold>. Then we use two 1 &#x000D7; 1 convolutional layers to obtain the attention weight of each channel, calculating through</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>w</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003B4;</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>w</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x000B7;</mml:mo><mml:mi>&#x003B4;</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>w</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold-italic"><mml:mi>g</mml:mi></mml:mstyle></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic><bold>w</bold></italic><sub><italic>Conv</italic>1</sub> and <italic><bold>w</bold></italic><sub><italic>Conv</italic>2</sub> are the weight parameters of two 1 &#x000D7; 1 convolutional layers, &#x003B4;(&#x000B7;) is the sigmoid activation function. Finally, the original input feature map is weighted by the weight vector,</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Z</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:mi>w</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02297;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>Z</mml:mtext></mml:mstyle><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x02297; means to multiply <italic><bold>w</bold></italic><sub><italic>att</italic></sub> and <bold>Z</bold> channel by channel.</p>
<p>In optimization, the channels that are highly relevant to lesion recognition are highlighted, which eliminates the information redundancy caused by the feature fusion of texture-biased stream and shape-biased stream, and selects the features conductive to lesion recognition, improving the robustness of the model.</p>
</sec>
<sec>
<title>3.5. Joint learning loss function and optimization</title>
<p>Due to the characteristics of the disease, training data often contains more benign lesions than malignant ones, resulting in insufficient attention given to malignant samples during network training and negatively impacting model optimization (Liu et al., <xref ref-type="bibr" rid="B24">2019a</xref>) and (Xu et al., <xref ref-type="bibr" rid="B45">2020</xref>). If the number of benign samples is forcibly reduced to balance the number of benign and malignant samples, it will lead to insufficient training data.</p>
<p>To address the problem of sample imbalance, we design the asymmetric loss function for medical image recognition with a large amount of negative samples and few positive samples. Different from the commonly used cross-entropy loss shown in Equation (5),</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>y</mml:mi><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>y</italic><sub> &#x02208; </sub>{0, 1} means the ground truth label of the sample, <italic>p</italic> &#x02208; (0, 1) is the predicted score, when <italic>p</italic>&#x0003E;0.5, the sample is predicted as the positive category, the asymmetric loss decouples the loss of positive and negative categories, reducing the impact of sample imbalance through asymmetric focusing and asymmetric probability transfer, for each sample, the new loss function for classification <inline-formula><mml:math id="M6"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>L</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is calculated through</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>L</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>y</mml:mi><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002B;</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B3;<sub>&#x0002B;</sub> and &#x003B3;<sub>&#x02212;</sub> are the exponential decay factors, the larger the value of the decay factor, the greater the attenuation effect. The adaptive weight factors <inline-formula><mml:math id="M8"><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002B;</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="M9"><mml:msup><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> are added to original cross-entropy loss function to asymmetrically scale the loss of positive samples and negative samples, which is better for the optimization in the case of unbalanced samples. We set &#x003B3;<sub>&#x0002B;</sub> &#x0003C; &#x003B3;<sub>&#x02212;</sub> to reduce the gradient of the negative samples, strengthening the attention of the model optimization to the positive samples.</p>
<p>In addition, with typical characteristics, some negative samples are easy to identify, to constrain the model to focus on hard samples, we add the probability transfer to the loss function, directly discarding samples which have a low predicted <italic>p</italic> value. The weight factor of <inline-formula><mml:math id="M10"><mml:mrow><mml:msub><mml:mi mathvariant="script">L</mml:mi><mml:mo>&#x02212;</mml:mo></mml:msub></mml:mrow></mml:math></inline-formula> is reconstructed with the transfer probability <italic>p</italic><sub><italic>t</italic></sub>, which is calculated by</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo class="qopname">max</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>-</mml:mo><mml:mi>&#x003C6;</mml:mi><mml:mo>,</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003C6; is the probability cutoff threshold, when the predicted <italic>p</italic> is lower than &#x003BC;, <italic>p</italic><sub><italic>t</italic></sub> is set to 0. The final asymmetric classification loss function is</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>L</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mi>y</mml:mi><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002B;</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:msubsup><mml:mo class="qopname">log</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>which enables the model to overcome the imbalance of samples in training, and focus on the difficult samples near the discrimination interface, enhancing the robustness of the trained model.</p>
<p>In the optimization of the shape-biased stream, we use <italic>L</italic>2 loss, which is the pixel-wise mean square error between the predicted mask <inline-formula><mml:math id="M13"><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>M</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> and the ground truth mask <bold>M</bold>, the shape loss <inline-formula><mml:math id="M14"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>H</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>H</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>M</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>-</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>M</mml:mtext></mml:mstyle><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In joint learning, texture feature encoder parameter <inline-formula><mml:math id="M16"><mml:mstyle mathvariant="bold"><mml:msubsup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msubsup></mml:mstyle></mml:math></inline-formula> is supervised by <italic>L</italic><sub><italic>CLS</italic></sub>, shape feature decoder parameter <inline-formula><mml:math id="M17"><mml:mstyle mathvariant="bold"><mml:msubsup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msubsup></mml:mstyle></mml:math></inline-formula> is supervised by <italic>L</italic><sub><italic>SHP</italic></sub>, shape feature encoder parameter <inline-formula><mml:math id="M18"><mml:mstyle mathvariant="bold"><mml:msubsup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msubsup></mml:mstyle></mml:math></inline-formula> is supervised by <italic>L</italic><sub><italic>CLS</italic></sub> and <italic>L</italic><sub><italic>SHP</italic></sub> to encourage learning shape features that are conductive to lesion classification. In summary, they are optimized by</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M19"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold-italic"><mml:msubsup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msubsup></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo class="qopname">arg&#x000A0;min</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>L</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E11"><label>(11)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold-italic"><mml:msubsup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msubsup></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo class="qopname">arg&#x000A0;min</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:munder></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B1;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>L</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B2;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>H</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E12"><label>(12)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold-italic"><mml:msubsup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow></mml:msubsup></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo class="qopname">arg&#x000A0;min</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold-italic"><mml:msub><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>H</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B1; and &#x003B2; is the scaling coefficient to balance <inline-formula><mml:math id="M22"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>L</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="M23"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>H</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, which is realized through the gradient scaling layer. Through the cooperative optimization of each module, the proposed method realizes texture and shape joint learning, improving the performance on shape-relied medical image recognition tasks.</p>
</sec>
</sec>
<sec id="s4">
<title>4. Experiments</title>
<sec>
<title>4.1. Experimental setup</title>
<sec>
<title>4.1.1. Data preparation</title>
<p>We use two medical image datasets to verify the effectiveness of the proposed method.</p>
<list list-type="bullet">
<list-item><p><bold>ISIC-2019:</bold> A public and commonly used dermoscopic image dataset for dermatological diagnose. According to the advice from dermatologists, the malignant melanoma is one of the most dangerous skin cancer, and the melanoma lesions have similar visual characteristics to nevus. Therefore, we focus on the melanoma and nevi recognition task on this dataset. We use 12,875 nevi images and 4,522 malignant melanoma images, of which 2,671 images have corresponding lesion mask labels.</p></list-item>
<list-item><p><bold>XJTU-MM:</bold> A skin pathological image dataset collected from the Second Affiliated Hospital of Xi&#x00027;an Jiaotong University(Xibei Hospital). It contains 9,098 images of RoI regions cropped from the whole slide histopathological images by pathologists, of which 2,170 images are malignant melanoma lesions and 6,928 images are benign nevus. And 726 of them have cell-wise masks labeled by pathologists.</p></list-item>
</list>
<p>The sample number of three datasets are shown in <xref ref-type="table" rid="T1">Table 1</xref>. Each dataset is divided into training set, validation set, and test set according to the ratio of 6:2:2, the images of malignant lesions are positive samples and the images of benign lesions are negative samples. Due to not all samples having the corresponding mask label, the shape-biased learning is only optimized when the input images have the corresponding mask labels.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Number of samples in each dataset.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Malignant</bold></th>
<th valign="top" align="center"><bold>Benign</bold></th>
<th valign="top" align="center"><bold>Total</bold></th>
<th valign="top" align="center"><bold>Mask label<sup>&#x0002A;</sup></bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">ISIC-2019</td>
<td valign="top" align="center">4,522</td>
<td valign="top" align="center">12,875</td>
<td valign="top" align="center">17,397</td>
<td valign="top" align="center">2,671</td>
</tr>
<tr>
<td valign="top" align="left">XJTU-MM</td>
<td valign="top" align="center">2,170</td>
<td valign="top" align="center">6,928</td>
<td valign="top" align="center">9,098</td>
<td valign="top" align="center">726</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><sup>&#x0002A;</sup>Due to not all samples having corresponding mask label, the shape-biased learning is only optimized when the input images have corresponding mask labels.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.1.2. Evaluation metrics</title>
<p>To quantitatively evaluate the performance of the model, we use accuracy(<italic>Acc</italic>.), precision(<italic>Pre</italic>.), recall(<italic>Rec</italic>.), and F1 score(<italic>F</italic>1) as evaluation metrics. They are calculated by</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M24"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mo>.</mml:mo><mml:mtext>&#x000A0;&#x000A0;=</mml:mtext><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>.</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mo>.</mml:mo><mml:mtext>&#x000A0;&#x000A0;=</mml:mtext><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mtext>&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x000D7;</mml:mo><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>.</mml:mo><mml:mo>&#x000D7;</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mo>.</mml:mo></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>.</mml:mo><mml:mo>+</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mo>.</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>TP</italic> (true positive) means the number of samples categorized to positive correctly, <italic>TN</italic> (true negative) means the number of samples categorized to negative correctly, <italic>FP</italic> (false positive) means the number of samples misclassified to malignant, <italic>FN</italic> (false negative) means the number of samples misclassified to negative. Higher accuracy reflects better overall performance of the model on all samples, higher precision means fewer malignant lesions are miss detected, and higher recall means higher sensitivity of the model to malignant lesions, F1 score is the combination of precision and recall. The four metrics provide a comprehensive evaluation of the medical image recognition models.</p>
</sec>
<sec>
<title>4.1.3. Implementation</title>
<p>In the proposed STNet-50, ResNet-50 is used as the baseline backbone of texture encoder and shape encoder, the shape feature decoder in the shape-biased stream is constructed using deconvolution operations and referring to the structure of ResNet-18. The texture encoder is pre-trained on ImageNet-1K. We implement the network using pytorch, opencv, scikit-learn and the libraries they depend on based on Python, and train the model on 2 RTX3090-24GB GPUs. All images are resized to 224 &#x000D7; 224, random rotation and random cropping are used for data augmentation. Batch size is set to 64, initial learning rate is set to 5<italic>e</italic>&#x02212;4, weight decay is set to 1<italic>e</italic>&#x02212;5, RMSprop (Hinton et al., <xref ref-type="bibr" rid="B19">2012</xref>) is used as the optimization algorithm and the momentum is set to 0.9. The exponential decay factors in asymmetric loss is set to &#x003BB;<sub>&#x0002B;</sub> &#x0003D; 1, &#x003BB;<sub>&#x02212;</sub> &#x0003D; 3.</p>
</sec>
</sec>
<sec>
<title>4.2. Comparison results</title>
<p>We compared the proposed method with some popular general vision models, including the ResNeSt (Zhang H. et al., <xref ref-type="bibr" rid="B51">2022</xref>), which is the latest iteration of ResNet, and ConvNeXt (Liu et al., <xref ref-type="bibr" rid="B26">2022</xref>), which is regarded as CNN for 2020s. We also added some models designed for specific medical image recognition tasks to the comparative experiment, including DeMAL-CNN (He et al., <xref ref-type="bibr" rid="B18">2022</xref>) for skin lesion classification in dermoscopy images, and MPMR (Zhang D. et al., <xref ref-type="bibr" rid="B49">2022</xref>), which is a multi-scale-feature-based melanoma recognition method in pathological images.</p>
<p>The results are shown in <xref ref-type="table" rid="T2">Table 2</xref>, which indicate that the proposed STNet outperforms compared algorithms on two datasets and on all evaluation metrics. ConvNeXt series models show generally better performance than ResNeSt-50 on two datasets, which confirms the progress from split-attention block to ConvNet block. DeMAL-CNN shows a similar ability to ConvNeXt on ISIC-2019 dataset, considering that it uses standard ResNet as the backbone, the framework design of DeMAL-CNN has considerable contributions to enhance the dermoscopic image feature representation. MPMR shows better performance than ConvNeXt, which indicates that enhancing multi-scale features is effective in skin pathology image recognition. In addition, in each series of models, the increase in network layers does not bring about significant performance improvements, it is difficult to significantly improve the recognition accuracy of the model simply by increasing the number of layers. Furthermore, in four evaluation metrics, precision and recall are obviously lower than accuracy, which is caused by the sample imbalance of malignant and benign samples. In this case, accuracy cannot comprehensively reflect the performance of the model, it is necessary to add other three metrics.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Quantitative results of the proposed method and the comparison method on ISIC-2019 and XJTU-MM datasets.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Model</bold></th>
<th valign="top" align="center"><bold><italic>Acc</italic>.&#x02191;</bold></th>
<th valign="top" align="center"><bold><italic>Pre</italic>.&#x02191;</bold></th>
<th valign="top" align="center"><bold><italic>Rec</italic>.&#x02191;</bold></th>
<th valign="top" align="center"><bold><italic>F</italic>1&#x02191;</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">ISIC-2019</td>
<td valign="top" align="center">ResNeSt-50</td>
<td valign="top" align="center">0.925</td>
<td valign="top" align="center">0.813</td>
<td valign="top" align="center">0.923</td>
<td valign="top" align="center">0.865</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">ResNeSt-101</td>
<td valign="top" align="center">0.927</td>
<td valign="top" align="center">0.816</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.869</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">ConvNeXt-S</td>
<td valign="top" align="center">0.949</td>
<td valign="top" align="center">0.858</td>
<td valign="top" align="center">0.964</td>
<td valign="top" align="center">0.908</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">ConvNeXt-B</td>
<td valign="top" align="center">0.957</td>
<td valign="top" align="center">0.881</td>
<td valign="top" align="center">0.965</td>
<td valign="top" align="center">0.921</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">DeMAL-50</td>
<td valign="top" align="center">0.952</td>
<td valign="top" align="center">0.864</td>
<td valign="top" align="center">0.967</td>
<td valign="top" align="center">0.913</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">DeMAL-101</td>
<td valign="top" align="center">0.954</td>
<td valign="top" align="center">0.878</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.915</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">STNet-50 (ours)</td>
<td valign="top" align="center">0.967</td>
<td valign="top" align="center">0.904</td>
<td valign="top" align="center">0.977</td>
<td valign="top" align="center">0.939</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">STNet-101 (ours)</td>
<td valign="top" align="center">0.971</td>
<td valign="top" align="center">0.916</td>
<td valign="top" align="center">0.978</td>
<td valign="top" align="center">0.946</td>
</tr>
<tr>
<td valign="top" align="left">XJTU-MM</td>
<td valign="top" align="center">ResNeSt-50</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.828</td>
<td valign="top" align="center">0.885</td>
<td valign="top" align="center">0.855</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">ResNeSt-101</td>
<td valign="top" align="center">0.933</td>
<td valign="top" align="center">0.846</td>
<td valign="top" align="center">0.880</td>
<td valign="top" align="center">0.863</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">ConvNeXt-S</td>
<td valign="top" align="center">0.945</td>
<td valign="top" align="center">0.868</td>
<td valign="top" align="center">0.908</td>
<td valign="top" align="center">0.887</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">ConvNeXt-B</td>
<td valign="top" align="center">0.946</td>
<td valign="top" align="center">0.875</td>
<td valign="top" align="center">0.901</td>
<td valign="top" align="center">0.888</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">MPMR-50</td>
<td valign="top" align="center">0.958</td>
<td valign="top" align="center">0.894</td>
<td valign="top" align="center">0.935</td>
<td valign="top" align="center">0.914</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">MPMR-101</td>
<td valign="top" align="center">0.961</td>
<td valign="top" align="center">0.910</td>
<td valign="top" align="center">0.929</td>
<td valign="top" align="center">0.919</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">STNet-50 (ours)</td>
<td valign="top" align="center">0.979</td>
<td valign="top" align="center">0.954</td>
<td valign="top" align="center">0.959</td>
<td valign="top" align="center">0.956</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">STNet-101 (ours)</td>
<td valign="top" align="center">0.985</td>
<td valign="top" align="center">0.963</td>
<td valign="top" align="center">0.972</td>
<td valign="top" align="center">0.968</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Some difficult samples in the test set of XJTU-MM dataset are visualized and shown in <xref ref-type="fig" rid="F6">Figure 6</xref>, where difficult samples mean the samples near the discriminant hyperplane. According to the results, The proposed STNet-50 correctly recognizes all of these samples. ResNeSt-50, ConvNeXt-S, and MPMR-50 all fail to recognition the first sample and the second sample, which contains rich irregular-shaped features. The fourth sample and the sixth sample have relatively distinct texture features distinct from melanoma, which is relatively easy to identify. The texture and feature joint learning enhances the shape feature representation, and the proposed asymmetric loss guides model to focus on difficult samples, so STNet has advantages on recognizing these difficult samples.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Visualized results of comparative experiment on XJTU-MM dataset. The green boxes mean correctly classified samples, the red boxes mean misclassified samples.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-17-1212049-g0006.tif"/>
</fig>
<p>In summary, the results of comparative experiments on ISIC-2019 and XJTU-MM datasets proves the effectiveness of our method.</p>
</sec>
<sec>
<title>4.3. Ablation analysis</title>
<p>To further study the contribution of each module in our method, we design ablation experiments to analyze the effect of pyramid-grouped convolution(PGC), deformable convolution(DC) and channel-attention-based feature fusion(CAFF) on model performance. we remove all of these modules from the proposed STNet-50 and use it as the baseline model (first row in <xref ref-type="table" rid="T3">Table 3</xref>). And then PGC, DC and CAFF are rejoined to baseline model one by one (row 2&#x02013;4 in <xref ref-type="table" rid="T3">Table 3</xref>). According to the results shown in <xref ref-type="table" rid="T3">Table 3</xref>, all the three modules bring performance improvement to model, especially in the increase of precision and recall. It indicates that PGC in the texture-biased stream and DC in shape-biased stream can both enhance the feature representation, and CAFF can select features that are more conducive to lesion identification. Additionally, these three modules are portable and can be plugged to other methods.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Results of ablation analysis of pyramid-grouped convolution(PGC), deformable convolution(DC) and channel-attention-based feature fusion(CAFF) on ISIC-2019 dataset.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left" colspan="3"><bold>Module</bold></th>
<th valign="top" align="center"><bold><italic>Acc</italic>.&#x02191;</bold></th>
<th valign="top" align="center"><bold><italic>Pre</italic>.&#x02191;</bold></th>
<th valign="top" align="center"><bold><italic>Rec</italic>.&#x02191;</bold></th>
<th valign="top" align="center"><bold><italic>F</italic>1&#x02191;</bold></th>
</tr>
<tr>
<th/>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><bold>PGC</bold></td>
<td valign="top" align="center"><bold>DC</bold></td>
<td valign="top" align="center"><bold>CAFF</bold></td>
<td/>
<td/>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">0.944</td>
<td valign="top" align="center">0.874</td>
<td valign="top" align="center">0.915</td>
<td valign="top" align="center">0.894</td>
</tr>
<tr>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">0.951</td>
<td valign="top" align="center">0.884</td>
<td valign="top" align="center">0.933</td>
<td valign="top" align="center">0.908</td>
</tr>
<tr>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">0.959</td>
<td valign="top" align="center">0.895</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.924</td>
</tr>
<tr>
<td valign="top" align="left">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">&#x02713;</td>
<td valign="top" align="center">0.967</td>
<td valign="top" align="center">0.904</td>
<td valign="top" align="center">0.977</td>
<td valign="top" align="center">0.939</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To further study the feature selection effect of CAFF in texture and shape feature fusion, we construct STNet-50 with CAFF and without CAFF respectively, and feed 500 malignant samples and 500 benign sample to them, for each sample, the feature vector in front of the classifier is input to t-SNE (Van der Maaten and Hinton, <xref ref-type="bibr" rid="B42">2008</xref>) manifold learning model to study the separability of the extracted features. Through t-SNE, the input feature vectors are transformed into two dimensions and visualized in <xref ref-type="fig" rid="F7">Figure 7</xref>. The comparison of <xref ref-type="fig" rid="F7">Figures 7A</xref>, <xref ref-type="fig" rid="F7">B</xref> show that the feature vector of the model with CAFF is more separable, which is conductive to classification. The results indicate that the introduction of CAFF module is effective to select features relevant to lesion recognition.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Visualized feature separability analysis through t-SNE. <bold>(A)</bold> Visualized result of STNet-50 without CAFF. <bold>(B)</bold> Visualized result of STNet-50 with CAFF. The feature vectors of STNet-50 with CAFF and STNet-50 without CAFF are transformed to two dimensions, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-17-1212049-g0007.tif"/>
</fig>
<p>Due to the available data is limited, to verify performance of the proposed model more rigorously, we conducted five-fold cross-validation on both ISIC-2019 and XJTU-MM datasets. Each dataset was divided into five mutually exclusive parts, with four used for training the STNet-50 model and one remaining part used for testing. Because of the sample imbalance problem, we use <italic>F</italic>1 score as the evaluation metric. The cross-validation results are shown in <xref ref-type="table" rid="T4">Table 4</xref>, STNet-50 shows consistent performance in each fold of the cross-validation, which proves the stability and reliability of the results.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Five-fold cross-validation results of the proposed STNet-50 model on ISIC-2019 and XJTU-MM datasets.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Datasets</bold></th>
<th valign="top" align="center" colspan="5"><italic><bold>F</bold></italic>1 <bold>score</bold> &#x02191;</th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>Fold 1</bold></th>
<th valign="top" align="center"><bold>Fold 2</bold></th>
<th valign="top" align="center"><bold>Fold 3</bold></th>
<th valign="top" align="center"><bold>Fold 4</bold></th>
<th valign="top" align="center"><bold>Fold 5</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">ISIC-2019</td>
<td valign="top" align="center">0.939</td>
<td valign="top" align="center">0.930</td>
<td valign="top" align="center">0.932</td>
<td valign="top" align="center">0.939</td>
<td valign="top" align="center">0.935</td>
</tr>
<tr>
<td valign="top" align="left">XJTU-MM</td>
<td valign="top" align="center">0.956</td>
<td valign="top" align="center">0.953</td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.953</td>
<td valign="top" align="center">0.952</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>4.4. Discussion on shape and texture joint learning framework</title>
<p>We propose the two-stream network for texture and shape joint learning, compared to single-stream network, an extra shape feature encoder is introduced. To analyze the contributions to performance improvements are provided by texture and shape joint learning or just the extra feature encoder, three control group models are designed for the comparative experiment. The first model uses the texture encoder only for feature extraction. The second model cascades the segmentation network and the classification network in the proposed method, the segmented lesion is used as the input of the classification network. The third model is constructed by removing the feature decoder of the shape-biased stream in our method, which is a two-stream network but without shape and texture joint learning. ISIC-2019 dataset is used for this experiment, the results are shown in <xref ref-type="table" rid="T5">Table 5</xref>, compared to the single-stream model, the cascade classification and segmentation model does not show obvious performance improvement and even have a performance drop on recall. It means that when the lesion mask labels are not sufficient, cascading the segmentation network and the classification network has limitation in solving weak shape representation problems. Two-stream network with joint learning shows better performance than that without joint learning, it indicates that the performance improvement of the proposed method is not simply brought by the extra shape feature encoder but by shape and texture joint learning, which proves the effectiveness of our method.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Experiments of discussion on shape and texture joint learning.</p></caption> 
<table frame="box" rules="all">
<thead>
<tr style="background-color:&#x00023;919498;color:&#x00023;ffffff">
<th valign="top" align="left"><bold>Backbone layers</bold></th>
<th valign="top" align="center"><bold>Structure</bold></th>
<th valign="top" align="center"><bold><italic>Acc</italic>.&#x02191;</bold></th>
<th valign="top" align="center"><bold><italic>Pre</italic>.&#x02191;</bold></th>
<th valign="top" align="center"><bold><italic>Rec</italic>.&#x02191;</bold></th>
<th valign="top" align="center"><bold><italic>F</italic>1&#x02191;</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">50</td>
<td valign="top" align="center">Single-stream<sup>a</sup></td>
<td valign="top" align="center">0.950</td>
<td valign="top" align="center">0.872</td>
<td valign="top" align="center">0.945</td>
<td valign="top" align="center">0.907</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Cascade Cls. and Seg.<sup>b</sup></td>
<td valign="top" align="center">0.950</td>
<td valign="top" align="center">0.886</td>
<td valign="top" align="center">0.928</td>
<td valign="top" align="center">0.907</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Two-stream without joint learning<sup>c</sup></td>
<td valign="top" align="center">0.960</td>
<td valign="top" align="center">0.909</td>
<td valign="top" align="center">0.939</td>
<td valign="top" align="center">0.924</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Two-stream with joint learning<sup>d</sup></td>
<td valign="top" align="center">0.967</td>
<td valign="top" align="center">0.904</td>
<td valign="top" align="center">0.977</td>
<td valign="top" align="center">0.939</td>
</tr>
<tr>
<td valign="top" align="left">101</td>
<td valign="top" align="center">Single-stream<sup>a</sup></td>
<td valign="top" align="center">0.952</td>
<td valign="top" align="center">0.877</td>
<td valign="top" align="center">0.950</td>
<td valign="top" align="center">0.912</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Cascade Cls. and Seg.<sup>b</sup></td>
<td valign="top" align="center">0.955</td>
<td valign="top" align="center">0.888</td>
<td valign="top" align="center">0.945</td>
<td valign="top" align="center">0.916</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Two-stream without joint learning<sup>c</sup></td>
<td valign="top" align="center">0.961</td>
<td valign="top" align="center">0.911</td>
<td valign="top" align="center">0.944</td>
<td valign="top" align="center">0.927</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Two-stream with joint learning<sup>d</sup></td>
<td valign="top" align="center">0.971</td>
<td valign="top" align="center">0.916</td>
<td valign="top" align="center">0.978</td>
<td valign="top" align="center">0.946</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><sup>a</sup>Single-stream: only use the texture encoder in the proposed method for feature extraction.</p>
<p><sup>b</sup>Cascade Cls. and Seg.: cascading segmentation network in front of classification network.</p>
<p><sup>c</sup>Two-stream without joint learning: removing the feature decoder in the shape-biased stream of our method.</p>
<p><sup>d</sup>Two-stream with joint learning: the proposed framework.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.5. Discussion on parameters of asymmetric loss</title>
<p>The asymmetric loss function in the proposed method is designed to address the sample imbalance problem, we use exponential decay factors &#x003B3;<sub>&#x0002B;</sub> and &#x003B3;<sub>&#x02212;</sub> to adjust the attention of the model to positive and negative classes. Due to in medical image datasets, malignant samples are usually much fewer than benign samples, &#x003B3;<sub>&#x02212;</sub> should achieve a stronger decay effect, so &#x003B3;<sub>&#x0002B;</sub> &#x0003C; &#x003B3;&#x02212;. To further study the effects of &#x003B3;<sub>&#x0002B;</sub> and &#x003B3;<sub>&#x02212;</sub> to model performance, we set &#x003B3;<sub>&#x0002B;</sub> &#x0003D; 1, and use different &#x003B3;<sub>&#x02212;</sub> to train the STNet-50 on ISIC-2019 dataset, the test results are shown in <xref ref-type="fig" rid="F8">Figure 8</xref>. Despite the model achieving the highest <italic>Pre</italic>. value When &#x003B3;<sub>&#x02212;</sub> &#x0003D; 2, taking into account the four metrics, the model has the best performance when &#x003B3;<sub>&#x02212;</sub> &#x0003D; 3. When &#x003B3;<sub>&#x02212;</sub> is too small, exponential decay is not enough to eliminate the impacts of sample imbalance. When &#x003B3;<sub>&#x02212;</sub> is too large, the effect of exponential decay is so strong that the model tends to ignore negative samples, and the performance of the model drops significantly. According to the results in <xref ref-type="fig" rid="F8">Figure 8</xref>, choosing an appropriate value of the exponential decay factor is important to train a good-performance model.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Variation of evaluation metrics with &#x003B3;<sub>&#x02212;</sub> when &#x003B3;<sub>&#x0002B;</sub> &#x0003D; 1. <italic>Acc</italic>., accuracy; <italic>Pre</italic>., precision; <italic>Rec</italic>., recall; <italic>F</italic>1, F1 score.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-17-1212049-g0008.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>5. Conclusion</title>
<p>In this paper, we propose the two-stream shape and texture joint learning network to address the weak shape feature representation problem of existing medical image recognition methods. According to the experiments on ISIC-2019 and XJTU-MM datasets, the proposed two-stream network is an effective method to combine texture and shape features. In addition, the proposed pyramid-grouped convolution enhances the texture feature representation, and deformable convolution enhances the shape feature representation. Furthermore, the channel-attention-based feature fusion module effectively eliminates redundant information and selects essential features. The asymmetric loss function addresses the problem of sample imbalance. The proposed method improves the model performance on shape-relied medical image recognition tasks, and provides support for computer-aided imaging diagnosis. Additionally, in our method, to enhance shape feature representation, an extra feature encoder is introduced, which increase the computation requirements, although the computation. Although inference speed is not the most critical concern in medical image analysis, we aim to enhance shape and texture feature representation by avoiding the use of additional encoders in future work, enhancing shape feature representation and texture feature representation within a single encoder.</p>
</sec>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>XW provided some ideas for this work. HH designed and implemented the models, ran the experiments, and wrote the manuscript. MenX analyzed the experimental data and visualized the results. SL helped write a part of the manuscript. DZ helped analyze the data and checked the manuscript writing. SD was in charge of project management. MeiX helps manage the project and provided advice for data analysis. All authors contributed to the article and approved the submitted version.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>This work was supported by the National Key Research and Development Program of China under Grant No. 2017YFA0700800, the Natural Science Basic Research Plan in Shaanxi Province of China under Grant No. 2022JM-324, the Social Science Foundation of Shaanxi Province of China under Grant No. 2021K014, and the Key Project of Shaanxi Province under Grant No. 2018ZDCXLGY-06-07.</p>
</sec>
<ack><p>Thanks to Longfei Zhu, an attending doctor from the Department of Dermatology of the Second Affiliated Hospital of Xi&#x00027;an Jiaotong University (Xibei Hospital), for his valuable advice on analysis of dermoscopic image data and pathology image data.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ahn</surname> <given-names>E.</given-names></name> <name><surname>Kim</surname> <given-names>J.</given-names></name> <name><surname>Bi</surname> <given-names>L.</given-names></name> <name><surname>Kumar</surname> <given-names>A.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Fulham</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Saliency-based lesion segmentation via background detection in dermoscopic images</article-title>. <source>IEEE J. Biomed. Health Inform.</source> <volume>21</volume>, <fpage>1685</fpage>&#x02013;<lpage>1693</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2017.2653179</pub-id><pub-id pub-id-type="pmid">28092585</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al-Osaimi</surname> <given-names>F. R.</given-names></name> <name><surname>Bennamoun</surname> <given-names>M.</given-names></name> <name><surname>Mian</surname> <given-names>A.</given-names></name></person-group> (<year>2011</year>). <article-title>Spatially optimized data-level fusion of texture and shape for face recognition</article-title>. <source>IEEE Trans. Image Process.</source> <volume>21</volume>, <fpage>859</fpage>&#x02013;<lpage>872</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2011.2165218</pub-id><pub-id pub-id-type="pmid">21859625</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Anantharatnasamy</surname> <given-names>P.</given-names></name> <name><surname>Sriskandaraja</surname> <given-names>K.</given-names></name> <name><surname>Nandakumar</surname> <given-names>V.</given-names></name> <name><surname>Deegalla</surname> <given-names>S.</given-names></name></person-group> (<year>2013</year>). <article-title>&#x0201C;Fusion of colour, shape and texture features for content based image retrieval,&#x0201D;</article-title> in <source>2013 8th International Conference on Computer Science &#x00026; Education</source> (<publisher-loc>Colombo</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>422</fpage>&#x02013;<lpage>427</lpage>.<pub-id pub-id-type="pmid">35496641</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Araujo</surname> <given-names>A.</given-names></name> <name><surname>Norris</surname> <given-names>W.</given-names></name> <name><surname>Sim</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>Computing receptive fields of convolutional neural networks</article-title>. <source>Distill</source> <volume>4</volume>, <fpage>e21</fpage>. <pub-id pub-id-type="doi">10.23915/distill.00021</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Celebi</surname> <given-names>M. E.</given-names></name> <name><surname>Codella</surname> <given-names>N.</given-names></name> <name><surname>Halpern</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>Dermoscopy image analysis: overview and future directions</article-title>. <source>IEEE J. Biomed. Health Inform.</source> <volume>23</volume>, <fpage>474</fpage>&#x02013;<lpage>478</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2019.2895803</pub-id><pub-id pub-id-type="pmid">30703051</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chan</surname> <given-names>H.-P.</given-names></name> <name><surname>Hadjiiski</surname> <given-names>L. M.</given-names></name> <name><surname>Samala</surname> <given-names>R. K.</given-names></name></person-group> (<year>2020</year>). <article-title>Computer-aided diagnosis in the era of deep learning</article-title>. <source>Med. Phys.</source> <volume>47</volume>, <fpage>e218</fpage>&#x02013;<lpage>e227</lpage>. <pub-id pub-id-type="doi">10.1002/mp.13764</pub-id><pub-id pub-id-type="pmid">32418340</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>H.</given-names></name></person-group> (<year>2017</year>). <article-title>Skin cancer reorganization and classification with deep neural network</article-title>. <source>arXiv preprint arXiv:1703.00534</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1703.00534</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>K.</given-names></name> <name><surname>Guo</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>C.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Enhanced breast lesion classification via knowledge guided cross-modal and semantic data augmentation,&#x0201D;</article-title> in <source>Medical Image Computing and Computer Assisted Intervention&#x02013;MICCAI 2021: 24th International Conference</source> (<publisher-loc>Strasbourg</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>53</fpage>&#x02013;<lpage>63</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chollet</surname> <given-names>F.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Xception: deep learning with depthwise separable convolutions,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Honolulu, HI</publisher-loc>), <fpage>1251</fpage>&#x02013;<lpage>1258</lpage>.<pub-id pub-id-type="pmid">36709517</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Dong</surname> <given-names>W.</given-names></name> <name><surname>Socher</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>L.-J.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name> <name><surname>Fei-Fei</surname> <given-names>L.</given-names></name></person-group> (<year>2009</year>). <article-title>&#x0201C;ImageNet: a large-scale hierarchical image database,&#x0201D;</article-title> in <source>2009 IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Miami, FL</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>248</fpage>&#x02013;<lpage>255</lpage>.<pub-id pub-id-type="pmid">26886976</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>H.</given-names></name> <name><surname>Xie</surname> <given-names>F.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Jiang</surname> <given-names>Z.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>Automatic segmentation of dermoscopy images using saliency combined with otsu threshold</article-title>. <source>Comput. Biol. Med.</source> <volume>85</volume>, <fpage>75</fpage>&#x02013;<lpage>85</lpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2017.03.025</pub-id><pub-id pub-id-type="pmid">28460258</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>C.</given-names></name> <name><surname>Arefan</surname> <given-names>D.</given-names></name> <name><surname>Panigrahy</surname> <given-names>A.</given-names></name> <name><surname>Zuley</surname> <given-names>M. L.</given-names></name> <name><surname>Wu</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Medical knowledge-guided deep learning for imbalanced medical image classification</article-title>. <source>arXiv preprint arXiv:2111.10620</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2111.10620</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Geirhos</surname> <given-names>R.</given-names></name> <name><surname>Rubisch</surname> <given-names>P.</given-names></name> <name><surname>Michaelis</surname> <given-names>C.</given-names></name> <name><surname>Bethge</surname> <given-names>M.</given-names></name> <name><surname>Wichmann</surname> <given-names>F. A.</given-names></name> <name><surname>Brendel</surname> <given-names>W.</given-names></name></person-group> (<year>2018</year>). <article-title>ImageNet-trained CNNs are biased towards texture; increasing shape bias improves accuracy and robustness</article-title>. <source>arXiv preprint arXiv:1811.12231</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1811.12231</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Rosing</surname> <given-names>T.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Depthwise convolution is all you need for learning multiple visual domains,&#x0201D;</article-title> in <source>Proceedings of the AAAI Conference on Artificial Intelligence</source> (<publisher-loc>Honolulu, HI</publisher-loc>), <fpage>8368</fpage>&#x02013;<lpage>8375</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Georgiou</surname> <given-names>T.</given-names></name> <name><surname>Lew</surname> <given-names>M. S.</given-names></name></person-group> (<year>2018</year>). <article-title>A review of semantic segmentation using deep neural networks</article-title>. <source>Int. J. Multimedia Inform. Retrieval</source> <volume>7</volume>, <fpage>87</fpage>&#x02013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1007/s13735-017-0141-z</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>H.</given-names></name> <name><surname>Du</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>D.</given-names></name> <name><surname>Long</surname> <given-names>H.</given-names></name> <name><surname>Guo</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Precise dental staging method through panoramic radiographs based on deep learning,&#x0201D;</article-title> in <source>2020 Chinese Automation Congress (CAC)</source> (<publisher-loc>Shanghai</publisher-loc>), <fpage>7406</fpage>&#x02013;<lpage>7411</lpage>. <pub-id pub-id-type="doi">10.1109/CAC51589.2020.9327719</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Gkioxari</surname> <given-names>G.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Mask r-CNN,&#x0201D;</article-title> in <source>Proceedings of the IEEE International Conference on Computer Vision</source> (<publisher-loc>Venice</publisher-loc>), <fpage>2961</fpage>&#x02013;<lpage>2969</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Zhao</surname> <given-names>S.</given-names></name> <name><surname>Yao</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>Deep metric attention learning for skin lesion classification in dermoscopy images</article-title>. <source>Complex Intell. Syst.</source> <volume>8</volume>, <fpage>1487</fpage>&#x02013;<lpage>1504</lpage>. <pub-id pub-id-type="doi">10.1007/s40747-021-00587-4</pub-id><pub-id pub-id-type="pmid">33445062</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G.</given-names></name> <name><surname>Srivastava</surname> <given-names>N.</given-names></name> <name><surname>Swersky</surname> <given-names>K.</given-names></name></person-group> (<year>2012</year>). <source>Neural Networks for Machine Learning Lecture 6a Overview of Mini-Batch Gradient Descent. Department of Computer Science, Toronto University, Toronto, ON, Canada</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.cs.toronto.edu/&#x0007E;tijmen/csc321/slides/lecture_slides_lec6.pdf">https://www.cs.toronto.edu/&#x0007E;tijmen/csc321/slides/lecture_slides_lec6.pdf</ext-link></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>G.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Van Der Maaten</surname> <given-names>L.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Densely connected convolutional networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Hawaii, HI</publisher-loc>), <fpage>4700</fpage>&#x02013;<lpage>4708</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jo</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>S. J.</given-names></name> <name><surname>Park</surname> <given-names>K. R.</given-names></name> <name><surname>Kim</surname> <given-names>I.-J.</given-names></name> <name><surname>Kim</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). <article-title>Detecting driver drowsiness using feature-level fusion and user-specific classification</article-title>. <source>Expert Syst. Appl.</source> <volume>41</volume>, <fpage>1139</fpage>&#x02013;<lpage>1152</lpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2013.07.108</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kotsia</surname> <given-names>I.</given-names></name> <name><surname>Zafeiriou</surname> <given-names>S.</given-names></name> <name><surname>Pitas</surname> <given-names>I.</given-names></name></person-group> (<year>2008</year>). <article-title>Texture and shape information fusion for facial expression and facial action unit recognition</article-title>. <source>Pattern Recogn.</source> <volume>41</volume>, <fpage>833</fpage>&#x02013;<lpage>851</lpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2007.06.026</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kurc</surname> <given-names>T.</given-names></name> <name><surname>Bakas</surname> <given-names>S.</given-names></name> <name><surname>Ren</surname> <given-names>X.</given-names></name> <name><surname>Bagari</surname> <given-names>A.</given-names></name> <name><surname>Momeni</surname> <given-names>A.</given-names></name> <name><surname>Huang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Segmentation and classification in digital pathology for glioma research: challenges and deep learning approaches</article-title>. <source>Front. Neurosci.</source> <volume>14</volume>, <fpage>27</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2020.00027</pub-id><pub-id pub-id-type="pmid">32153349</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>T.</given-names></name> <name><surname>Fan</surname> <given-names>W.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name></person-group> (<year>2019a</year>). <article-title>A hybrid machine learning approach to cerebral stroke prediction based on imbalanced medical dataset</article-title>. <source>Artif. Intell. Med.</source> <volume>101</volume>, <fpage>101723</fpage>. <pub-id pub-id-type="doi">10.1016/j.artmed.2019.101723</pub-id><pub-id pub-id-type="pmid">31813482</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>T.</given-names></name> <name><surname>Guo</surname> <given-names>Q.</given-names></name> <name><surname>Lian</surname> <given-names>C.</given-names></name> <name><surname>Ren</surname> <given-names>X.</given-names></name> <name><surname>Liang</surname> <given-names>S.</given-names></name> <name><surname>Yu</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2019b</year>). <article-title>Automated detection and classification of thyroid nodules in ultrasound images using clinical-knowledge-guided convolutional neural networks</article-title>. <source>Med. Image Anal.</source> <volume>58</volume>, <fpage>101555</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2019.101555</pub-id><pub-id pub-id-type="pmid">31520984</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Mao</surname> <given-names>H.</given-names></name> <name><surname>Wu</surname> <given-names>C.-Y.</given-names></name> <name><surname>Feichtenhofer</surname> <given-names>C.</given-names></name> <name><surname>Darrell</surname> <given-names>T.</given-names></name> <name><surname>Xie</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;A ConvNet for the 2020s,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>New Orleans, LA</publisher-loc>), <fpage>11976</fpage>&#x02013;<lpage>11986</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Long</surname> <given-names>J.</given-names></name> <name><surname>Shelhamer</surname> <given-names>E.</given-names></name> <name><surname>Darrell</surname> <given-names>T.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Fully convolutional networks for semantic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Boston, MA</publisher-loc>), <fpage>3431</fpage>&#x02013;<lpage>3440</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>S.</given-names></name> <name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>N.</given-names></name></person-group> (<year>2023</year>). <article-title>PKRT-Net: prior knowledge-based relation transformer network for optic cup and disc segmentation</article-title>. <source>Neurocomputing</source> <volume>538</volume>, <fpage>126183</fpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2023.03.044</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>Z.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>Q.</given-names></name></person-group> (<year>2017</year>). <article-title>Face image retrieval based on shape and texture feature fusion</article-title>. <source>Comput. Visual Media</source> <volume>3</volume>, <fpage>359</fpage>&#x02013;<lpage>368</lpage>. <pub-id pub-id-type="doi">10.1007/s41095-017-0091-7</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Urtasun</surname> <given-names>R.</given-names></name> <name><surname>Zemel</surname> <given-names>R.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Understanding the effective receptive field in deep convolutional neural networks,&#x0201D;</article-title> in <source>30th Conference on Neural Information Processing Systems (NIPS 2016)</source>, eds D. Lee, M. Sugiyama, U. Luxburg, I. Guyon and R. Garnett [Barcelona: Neural Information Processing Systems Foundation, Inc. (NeurIPS)], <fpage>4898</fpage>&#x02013;<lpage>4906</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>N.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Zheng</surname> <given-names>H.-T.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;ShuffleNet V2: practical guidelines for efficient CNN architecture design,&#x0201D;</article-title> in <source>Proceedings of the European Conference on Computer Vision (ECCV)</source> (<publisher-loc>Munich</publisher-loc>), <fpage>116</fpage>&#x02013;<lpage>131</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ni</surname> <given-names>D.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Qin</surname> <given-names>J.</given-names></name> <name><surname>Ouyang</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>&#x0201C;Learning based automatic head detection and measurement from fetal ultrasound images via prior knowledge and imaging parameters,&#x0201D;</article-title> in <source>2013 IEEE 10th International Symposium on Biomedical Imaging</source> (<publisher-loc>San Francisco, CA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>772</fpage>&#x02013;<lpage>775</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oktay</surname> <given-names>O.</given-names></name> <name><surname>Schlemper</surname> <given-names>J.</given-names></name> <name><surname>Folgoc</surname> <given-names>L. L.</given-names></name> <name><surname>Lee</surname> <given-names>M.</given-names></name> <name><surname>Heinrich</surname> <given-names>M.</given-names></name> <name><surname>Misawa</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Attention U-Net: learning where to look for the pancreas</article-title>. <source>arXiv preprint arXiv:1804.03999</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1804.03999</pub-id><pub-id pub-id-type="pmid">35474556</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname> <given-names>L.</given-names></name> <name><surname>Cai</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>N.</given-names></name> <name><surname>Yang</surname> <given-names>L.</given-names></name> <name><surname>Zheng</surname> <given-names>S.</given-names></name> <name><surname>Huang</surname> <given-names>L.</given-names></name></person-group> (<year>2022</year>). <article-title>A two-stage network with prior knowledge guidance for medullary thyroid carcinoma recognition in ultrasound images</article-title>. <source>Med. Phys.</source> <volume>49</volume>, <fpage>2413</fpage>&#x02013;<lpage>2426</lpage>. <pub-id pub-id-type="doi">10.1002/mp.15492</pub-id><pub-id pub-id-type="pmid">35103313</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ronneberger</surname> <given-names>O.</given-names></name> <name><surname>Fischer</surname> <given-names>P.</given-names></name> <name><surname>Brox</surname> <given-names>T.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;U-Net: convolutional networks for biomedical image segmentation,&#x0201D;</article-title> in <source>Medical Image Computing and Computer-Assisted Intervention&#x02013;MICCAI 2015: 18th International Conference</source> (<publisher-loc>Munich</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>234</fpage>&#x02013;<lpage>241</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rotemberg</surname> <given-names>V.</given-names></name> <name><surname>Kurtansky</surname> <given-names>N.</given-names></name> <name><surname>Betz-Stablein</surname> <given-names>B.</given-names></name> <name><surname>Caffery</surname> <given-names>L.</given-names></name> <name><surname>Chousakos</surname> <given-names>E.</given-names></name> <name><surname>Codella</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>A patient-centric dataset of images and metadata for identifying melanomas using clinical context</article-title>. <source>Sci. Data</source> <volume>8</volume>, <fpage>34</fpage>. <pub-id pub-id-type="doi">10.1038/s41597-021-00815-z</pub-id><pub-id pub-id-type="pmid">33727560</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Russakovsky</surname> <given-names>O.</given-names></name> <name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Su</surname> <given-names>H.</given-names></name> <name><surname>Krause</surname> <given-names>J.</given-names></name> <name><surname>Satheesh</surname> <given-names>S.</given-names></name> <name><surname>Ma</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>ImageNet large scale visual recognition challenge</article-title>. <source>Int. J. Comput. Vision</source> <volume>115</volume>, <fpage>211</fpage>&#x02013;<lpage>252</lpage>. <pub-id pub-id-type="doi">10.1007/s11263-015-0816-y</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>D.</given-names></name> <name><surname>Wu</surname> <given-names>G.</given-names></name> <name><surname>Suk</surname> <given-names>H.-I.</given-names></name></person-group> (<year>2017</year>). <article-title>Deep learning in medical image analysis</article-title>. <source>Annu. Rev. Biomed. Eng.</source> <volume>19</volume>, <fpage>221</fpage>&#x02013;<lpage>248</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-bioeng-071516-044442</pub-id><pub-id pub-id-type="pmid">28301734</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Qiang</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>X.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name> <name><surname>Hao</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Knowledge-guided synthetic medical image adversarial augmentation for ultrasonography thyroid nodule classification</article-title>. <source>Comput. Methods Prog. Biomed.</source> <volume>196</volume>, <fpage>105611</fpage>. <pub-id pub-id-type="doi">10.1016/j.cmpb.2020.105611</pub-id><pub-id pub-id-type="pmid">32650266</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sumathi</surname> <given-names>C.</given-names></name> <name><surname>Kumar</surname> <given-names>A. S.</given-names></name></person-group> (<year>2012</year>). <article-title>Edge and texture fusion for plant leaf classification</article-title>. <source>Int. J. Comput. Sci. Telecommun.</source> <volume>3</volume>, <fpage>6</fpage>&#x02013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>M.</given-names></name> <name><surname>Le</surname> <given-names>Q. V.</given-names></name></person-group> (<year>2019</year>). <article-title>Mixconv: Mixed depthwise convolutional kernels</article-title>. <source>arXiv preprint arXiv:1907.09595</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1907.09595</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van der Maaten</surname> <given-names>L.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2008</year>). <article-title>Visualizing data using t-SNE</article-title>. <source>J. Mach. Learn. Res.</source> <volume>9</volume>, <fpage>2579</fpage>&#x02013;<lpage>2605</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xie</surname> <given-names>S.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name> <name><surname>Tu</surname> <given-names>Z.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Aggregated residual transformations for deep neural networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Hawaii, HI</publisher-loc>), <fpage>1492</fpage>&#x02013;<lpage>1500</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiong</surname> <given-names>L.</given-names></name> <name><surname>Zheng</surname> <given-names>N.</given-names></name> <name><surname>You</surname> <given-names>Q.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name></person-group> (<year>2007</year>). <article-title>&#x0201C;Facial expression sequence synthesis based on shape and texture fusion model,&#x0201D;</article-title> in <source>2007 IEEE International Conference on Image Processing</source> (<publisher-loc>San Antonio, TX</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4</fpage>&#x02013;<lpage>473</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Shen</surname> <given-names>D.</given-names></name> <name><surname>Nie</surname> <given-names>T.</given-names></name> <name><surname>Kou</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>A hybrid sampling algorithm combining m-smote and ENN based on random forest for medical imbalanced data</article-title>. <source>J. Biomed. Inform.</source> <volume>107</volume>, <fpage>103465</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2020.103465</pub-id><pub-id pub-id-type="pmid">32512209</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yanase</surname> <given-names>J.</given-names></name> <name><surname>Triantaphyllou</surname> <given-names>E.</given-names></name></person-group> (<year>2019</year>). <article-title>A systematic survey of computer-aided diagnosis in medicine: past and present developments</article-title>. <source>Expert Syst. Appl.</source> <volume>138</volume>, <fpage>112821</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2019.112821</pub-id></citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>W.</given-names></name> <name><surname>Dong</surname> <given-names>Y.</given-names></name> <name><surname>Du</surname> <given-names>Q.</given-names></name> <name><surname>Qiang</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>K.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Integrate domain knowledge in training multi-task cascade deep learning model for benign&#x02013;malignant thyroid nodule classification on ultrasound images</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>98</volume>, <fpage>104064</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2020.104064</pub-id></citation>
</ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Xie</surname> <given-names>F.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Skin lesion classification based on two-modal images using a multi-scale fully-shared fusion network</article-title>. <source>Comput. Methods Prog. Biomed.</source> <volume>229</volume>, <fpage>107315</fpage>. <pub-id pub-id-type="doi">10.1016/j.cmpb.2022.107315</pub-id><pub-id pub-id-type="pmid">36586177</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>D.</given-names></name> <name><surname>Han</surname> <given-names>H.</given-names></name> <name><surname>Du</surname> <given-names>S.</given-names></name> <name><surname>Zhu</surname> <given-names>L.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>MPMR: multi-scale feature and probability map for melanoma recognition</article-title>. <source>Front. Med.</source> <volume>8</volume>, <fpage>775587</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2021.775587</pub-id><pub-id pub-id-type="pmid">35071264</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>D.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Du</surname> <given-names>S.</given-names></name> <name><surname>Han</surname> <given-names>H.</given-names></name> <name><surname>Ge</surname> <given-names>Y.</given-names></name> <name><surname>Zhu</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Coarse-to-fine feature representation based on deformable partition attention for melanoma identification</article-title>. <source>Pattern Recogn.</source> <volume>136</volume>, <fpage>109247</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2022.109247</pub-id></citation>
</ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>&#x0201C;Resnest: split-attention networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, <fpage>2736</fpage>&#x02013;<lpage>2746</lpage>.</citation>
</ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Kosov</surname> <given-names>S.</given-names></name> <name><surname>Grzegorzek</surname> <given-names>M.</given-names></name> <name><surname>Shirahama</surname> <given-names>K.</given-names></name> <name><surname>Jiang</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>LCU-Net: a novel low-cost u-net for environmental microorganism image segmentation</article-title>. <source>Pattern Recogn.</source> <volume>115</volume>, <fpage>107885</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2021.107885</pub-id></citation>
</ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Tao</surname> <given-names>C.</given-names></name></person-group> (<year>2018a</year>). <article-title>Towards improving diagnosis of skin diseases by combining deep neural network and human knowledge</article-title>. <source>BMC Med. Inform. Decis. Mak.</source> <volume>18</volume>, <fpage>59</fpage>. <pub-id pub-id-type="doi">10.1186/s12911-018-0631-9</pub-id><pub-id pub-id-type="pmid">30066649</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Zhou</surname> <given-names>X.</given-names></name> <name><surname>Lin</surname> <given-names>M.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2018b</year>). <article-title>&#x0201C;ShuffleNet: an extremely efficient convolutional neural network for mobile devices,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>), <fpage>6848</fpage>&#x02013;<lpage>6856</lpage>.</citation>
</ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>P.</given-names></name> <name><surname>Yuan</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Texture and shape biased two-stream networks for clothing classification and attribute recognition,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Seattle, WA</publisher-loc>), <fpage>13538</fpage>&#x02013;<lpage>13547</lpage>.</citation>
</ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Z.</given-names></name> <name><surname>Rahman Siddiquee</surname> <given-names>M. M.</given-names></name> <name><surname>Tajbakhsh</surname> <given-names>N.</given-names></name> <name><surname>Liang</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;UNet&#x0002B;&#x0002B;: a nested U-Net architecture for medical image segmentation,&#x0201D;</article-title> in <source>Deep Learning in Medical Image Analysis and Multimodal Learning for Clinical Decision Support: 4th International Workshop, DLMIA 2018, and 8th International Workshop, ML-CDS 2018</source> (<publisher-loc>Granada</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>3</fpage>&#x02013;<lpage>11</lpage>.<pub-id pub-id-type="pmid">32613207</pub-id></citation></ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Z.</given-names></name> <name><surname>Zhao</surname> <given-names>C.</given-names></name> <name><surname>Qiao</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>M.</given-names></name> <name><surname>Guo</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Q.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Rating: medical knowledge-guided rheumatoid arthritis assessment from multimodal ultrasound images via deep learning</article-title>. <source>Patterns</source> <volume>3</volume>, <fpage>100592</fpage>. <pub-id pub-id-type="doi">10.1016/j.patter.2022.100592</pub-id><pub-id pub-id-type="pmid">36277816</pub-id></citation></ref>
</ref-list> 
</back>
</article>